{"as_of":"2026-08-11T20:38:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:180f8b1d4256e646595e12d8403d8e8ec97a4bfada53875952c02dc3dbf0444f","coverage":[{"denominator":102,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-10T10:34:01.517256Z","state":"measured"},{"denominator":101,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":101,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-11T06:34:44.6726+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-14T22:09:28.433165Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-14T22:09:30.691155Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"cited_work":{"arxiv_id":"2501.16811","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.16811","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Not every patch is needed: Towards a more efficient and effective backbone for video-based person re-identification","venue":null,"work_id":"904e406e-02f1-423e-b08f-fb3f9710ad56","year":2025},"citing_paper":{"arxiv_id":"2603.27222","last_updated":"2026-04-10T16:16:05Z","snapshot_observed_at":"2026-07-06T22:50:55.897076Z","submitted_at":"2026-03-28T10:29:07Z","title":"HD-VGGT: High-Resolution Visual Geometry Transformer","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-14T22:09:28.433165Z"},"links":{"cited_paper":"/paper/2501.16811","citing_paper":"/paper/2603.27222"},"observation_digest":"sha256:93c0e29955174b2a7e45691b7b171030a6e9828dbeda3fb711592f3e8e2fc524","observation_id":"3b8a5e8e-1997-4d87-8d47-ed4499b58348","resolution":{"observed_at":"2026-05-14T22:09:30.695324Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2501.16811/citation-record","integrity":"/paper/2501.16811/integrity","json":"/paper/2501.16811/citation-record.json","paper":"/paper/2501.16811"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:00.402053Z","title":"Spatio-temporal repre- sentation factorization for video-based person re-identification","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.402053Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:72a98d3cca55ef6bde91b8635c7b43e3db0e425297e9bf9af2f8734b6f260ee2","observation_id":"be51c658-c095-4322-ab77-4ae19d16f998","resolution":{"observed_at":"2026-08-10T10:34:00.402053Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:00.411694Z","title":"Salient-to-broad transition for video person re-identification","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.411694Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:19b8f57c5d8c4030cc357e2a72202ba0714d8e9c391b1f6b4303819e726fef7f","observation_id":"5331d4ee-28fd-4796-873e-5450b91e498e","resolution":{"observed_at":"2026-08-10T10:34:00.411694Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:00.420728Z","title":"Event-guided person re- identification via sparse-dense complementary learning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.420728Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:40bb70f66b14e4760347f5de0c4e29f495e68ad6034e3e9ad231ca58266b8959","observation_id":"56d5aa24-dbd6-4916-9944-84dbd68fdb78","resolution":{"observed_at":"2026-08-10T10:34:00.420728Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:00.436696Z","title":"End-to-end object detection with transformers","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.436696Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:5303425058fe254266dcaa7930095e89d6a19e36e240f9eb254d00a27c99d9e5","observation_id":"90154ed6-1fe0-49d7-a9df-44bf565a5cbd","resolution":{"observed_at":"2026-08-10T10:34:00.436696Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:00.445164Z","title":"Video person re-identification with competitive snippet- similarity aggregation and co-attentive snippet embedding","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.445164Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:f25a3f9060ac37c7a24d6e4e20409ca6e33f7f59e2688142b7bb072f7d04c6b9","observation_id":"ecdd2d22-8b68-49ec-a390-b9c5ee31bea8","resolution":{"observed_at":"2026-08-10T10:34:00.445164Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:00.455961Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.455961Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:683779d505de742f3262e5df0f56add630cab4c3fbf94ed85168de4ff30c5215","observation_id":"65b52554-710d-4ba4-8361-254266588c79","resolution":{"observed_at":"2026-08-10T10:34:00.455961Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:00.467257Z","title":"Diffrate: Differentiable compression rate for efficient vision transformers","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.467257Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:5a610d18e0d9f3a6f023beb8ef65569693df13f8436b0802f27f4aa73ae6516c","observation_id":"433ae3d1-7382-441d-aca6-116b6032def2","resolution":{"observed_at":"2026-08-10T10:34:00.467257Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:00.487323Z","title":"Reality3dsketch: Rapid 3d modeling of objects from single freehand sketches","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.487323Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:b4b980a8e91b8080ef9b4e2e2ef520e234944adb3ab63406c29690396c54a096","observation_id":"00f87cd1-c2b5-4f55-b84a-5f9968001dca","resolution":{"observed_at":"2026-08-10T10:34:00.487323Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:00.495338Z","title":"Abd-net: Attentive but diverse person re-identification","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.495338Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:c03adc1aa9517845c7b6f022ab0f34f550886d7a0663237b36144dc135a7b649","observation_id":"90c2b3f2-02d5-4622-905b-d27dad233155","resolution":{"observed_at":"2026-08-10T10:34:00.495338Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04435","last_updated":"2023-12-07T16:57:38Z","snapshot_observed_at":"2026-07-06T16:58:23.647152Z","submitted_at":"2023-12-07T16:57:38Z","title":"Deep3DSketch: 3D modeling from Free-hand Sketches with View- and Structural-Aware Adversarial Training","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04435","snapshot_observed_at":"2026-08-10T10:34:00.504460Z","title":"Deep3dsketch: 3d modeling from free-hand sketches with view-and structural-aware adversarial training","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.504460Z"},"links":{"cited_paper":"/paper/2312.04435","citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:846530cd6ca487fff8cfbf061a47e6b4fb65dab2ee666b50efc022030356765f","observation_id":"e96a7da9-8e65-4853-bd8f-95f28a4ac6fb","resolution":{"observed_at":"2026-08-10T10:34:00.504460Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04579","last_updated":"2024-08-10T11:20:52Z","snapshot_observed_at":"2026-07-06T18:58:30.942582Z","submitted_at":"2024-08-08T16:40:15Z","title":"SAM2-Adapter: Evaluating & Adapting Segment Anything 2 in Downstream Tasks: Camouflage, Shadow, Medical Image Segmentation, and More","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.04579","snapshot_observed_at":"2026-08-10T10:34:00.512010Z","title":"Sam2- adapter: Evaluating & adapting segment anything 2 in downstream tasks: Camouflage, shadow, medical image segmentation, and more","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.512010Z"},"links":{"cited_paper":"/paper/2408.04579","citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:2905e3327f45391bd8261395590f93fe340a69cf84c6aff0fb0afdc2709f5522","observation_id":"66d5905d-e82f-4c19-ba91-494ee22e5e33","resolution":{"observed_at":"2026-08-10T10:34:00.512010Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.19326","last_updated":"2024-05-29T17:56:07Z","snapshot_observed_at":"2026-07-06T18:22:10.094519Z","submitted_at":"2024-05-29T17:56:07Z","title":"Reasoning3D -- Grounding and Reasoning in 3D: Fine-Grained Zero-Shot Open-Vocabulary 3D Reasoning Part Segmentation via Large Vision-Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.19326","snapshot_observed_at":"2026-08-10T10:34:00.519569Z","title":"Reasoning3d– grounding and reasoning in 3d: Fine-grained zero-shot open-vocabulary 3d reasoning part segmentation via large vision-language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.519569Z"},"links":{"cited_paper":"/paper/2405.19326","citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:d981051544191972363dc7dc96ebfd8e2409e6fcebcca36464e80207c3fec231","observation_id":"1c5017d8-2451-4f00-bb52-7ab188e989c6","resolution":{"observed_at":"2026-08-10T10:34:00.519569Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:00.530575Z","title":"Sam-adapter: Adapting segment anything in underperformed scenes","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.530575Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:4a85318279e22560c3b4b3d0a19796053102c134f880b8b845dc49b0103eea8c","observation_id":"eef637cf-3f38-4267-ae1b-5f4386982dc5","resolution":{"observed_at":"2026-08-10T10:34:00.530575Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:00.546756Z","title":"Masked-attention mask transformer for universal image segmentation","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.546756Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:c32b583ac253b6766c18ff2b0ed6ac950a347ccf8df8edf9c6f6e280012c0da0","observation_id":"ebd7810f-a237-4ada-9c16-2cfd45f69fbf","resolution":{"observed_at":"2026-08-10T10:34:00.546756Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:00.554536Z","title":"Towards accurate post-training quantization for vision transformer","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.554536Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:1e348417106b59522f9bbe8ab74a19be555cfa5c91b5790ac178d71e284b275e","observation_id":"3c1a51ba-db07-409b-9872-9cac7dab837b","resolution":{"observed_at":"2026-08-10T10:34:00.554536Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2010.11929","last_updated":"2021-06-03T13:08:56Z","snapshot_observed_at":"2026-08-10T01:12:16.468283Z","submitted_at":"2020-10-22T17:55:59Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.11929","snapshot_observed_at":"2026-08-10T10:34:00.564210Z","title":"An image is worth 16x16 words: Transformers for image recognition at scale","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.564210Z"},"links":{"cited_paper":"/paper/2010.11929","citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:a8abdbcfe1c6ca429f0743247096e5c5f6dd0dcb662e1c0b9d81b7ded9967b34","observation_id":"0f9310be-8449-4ff2-ab3d-872ad65edcf0","resolution":{"observed_at":"2026-08-10T10:34:00.564210Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:00.572592Z","title":"Eventful transformers: Leveraging temporal redundancy in vision transformers","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.572592Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:4e79a6c7282b3b2a03674eaaf6b87da09b7338953290be5fb477ecc45282115c","observation_id":"f6aab266-ce41-40d9-85be-40e8511a2972","resolution":{"observed_at":"2026-08-10T10:34:00.572592Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:00.582628Z","title":"Video- based person re-identification with spatial and temporal memory net- works","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.582628Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:58318c3dca2eb4843cc15e0cb2bb6d5006fa706370a51ccbd1b45106b167f8f2","observation_id":"54c3c877-9305-460e-8706-b375752924b5","resolution":{"observed_at":"2026-08-10T10:34:00.582628Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:00.608030Z","title":"Motion adaptive pose estimation from compressed videos","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.608030Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:d8b28850523c73b812fe2d2e6ae248ef19d902f2bc53cd81f3901ca283a679eb","observation_id":"656a2bd3-e1a5-49af-a9fc-74650e0963de","resolution":{"observed_at":"2026-08-10T10:34:00.608030Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:00.625173Z","title":"Sta: Spatial-temporal attention for large-scale video-based person re- identification","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.625173Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:9d11b6361f20e37d7e3b3de880eba41062f704bfdc9144bc1f151f14490d7784","observation_id":"0480bfbf-e940-47fd-9373-93d6ac9a172c","resolution":{"observed_at":"2026-08-10T10:34:00.625173Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03768","last_updated":"2023-04-07T17:59:58Z","snapshot_observed_at":"2026-08-10T16:17:02.239899Z","submitted_at":"2023-04-07T17:59:58Z","title":"SparseFormer: Sparse Visual Recognition via Limited Latent Tokens","version":1},"cited_work":{"arxiv_id":"2304.03768","doi":null,"metadata_source":"pith","pith_arxiv_id":"2304.03768","snapshot_observed_at":"2026-08-10T10:34:01.840079Z","title":"SparseFormer: Sparse Visual Recognition via Limited Latent Tokens","venue":"cs.CV","work_id":"ae6e2dba-599f-430b-8868-746d37e3c67b","year":2023},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.630751Z"},"links":{"cited_paper":"/paper/2304.03768","citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:de92c1b28b98199ea278508042d7087476a2d3eb284d11cfb45fb2fd5019a448","observation_id":"f5749dd6-6132-4cd6-9556-a1ecc125c45d","resolution":{"observed_at":"2026-08-10T10:34:01.847747Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:00.639133Z","title":"Appearance-preserving 3d convolution for video-based person re-identification","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.639133Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:98e573bb408992762835108ca6f142a357efa307d725bbdf8d4a224e806e59be","observation_id":"3f4a3073-e7f5-4911-b942-92c7be60e771","resolution":{"observed_at":"2026-08-10T10:34:00.639133Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:00.646378Z","title":"Flatten transformer: Vision transformer using focused linear attention","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.646378Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:ae09ebc3c10eeada726ef5f7742aaa9ab5b85363ad04ffd3365127f40a501def","observation_id":"d4c22ab1-99fb-4af4-b6f4-033f63cdd966","resolution":{"observed_at":"2026-08-10T10:34:00.646378Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:00.655144Z","title":"Deep residual learning for image recognition","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.655144Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:0f786126130b6a90b212e90b367db0c808a8ea5c4ef44edb9f76c5efaf11c452","observation_id":"acb84bea-0f12-432f-985d-9a08d8dfd446","resolution":{"observed_at":"2026-08-10T10:34:00.655144Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:00.661142Z","title":"Transreid: Transformer-based object re-identification","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.661142Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:734c87ae203dbf8a45b433777be97a2241c1bd1fe905600555f6f10ed4401ae4","observation_id":"4d63f547-094d-41fa-b1c5-28796c2e2061","resolution":{"observed_at":"2026-08-10T10:34:00.661142Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:00.679988Z","title":"Bicnet-tks: Learning efficient spatial-temporal representation for video person re-identification","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.679988Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:4087528102c8354c9ba383f843e2915b804326680f2febf1d782dd175fa470a4","observation_id":"c1442edd-9dfb-4961-9f3d-beb9a487176d","resolution":{"observed_at":"2026-08-10T10:34:00.679988Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:00.687909Z","title":"Temporal complementary learning for video person re-identification","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.687909Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:4ceb23b90caecc3665e27462facba7e50002ed5d0e7cf58651d484452a0ecb19","observation_id":"46249f9c-6e96-4396-a7bc-f6ea6c22c8d4","resolution":{"observed_at":"2026-08-10T10:34:00.687909Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:04.655326Z","title":"Vrstc: Occlusion-free video person re-identification","venue":null,"work_id":"72d51492-101d-4257-81d6-9b411e35f401","year":2019},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.693101Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:017789e7a63f63eae5f12fb1575df32884119ce14f4178aa5cecd857cf4297e8","observation_id":"72827808-4fda-42c7-86af-a46a71c96b7a","resolution":{"observed_at":"2026-08-10T10:34:04.661983Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:04.610683Z","title":"Orthogonal transformer: An efficient vision transformer backbone with token orthogonalization","venue":null,"work_id":"02015b9f-4f93-4910-b3a4-4a1d49b177a1","year":2022},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.698633Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:f3801bd584f77ec8c07d34d6234ec7a2f6a349269ec8290846d6307431c331fc","observation_id":"9bc71934-e46b-4d09-a639-34b90dd9c969","resolution":{"observed_at":"2026-08-10T10:34:04.620085Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:04.570634Z","title":"Reasoning and tuning: Graph attention network for occluded person re- identification","venue":null,"work_id":"e250a7c0-26af-451d-a1d5-3e850570a2c6","year":2023},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.703468Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:eac401a4ad9a01fc7ef2d3ac87ff762df7b251819cc22bb01af47debc7a2236b","observation_id":"1517c3fa-2eb8-4df8-88d9-3fbec8c5e1a6","resolution":{"observed_at":"2026-08-10T10:34:04.581239Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:04.540652Z","title":"En- hancing person re-identification performance through in vivo learning","venue":null,"work_id":"0aa1fa18-9c4b-4cd5-b10b-a959501a793d","year":2023},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.708947Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:3048b9c19950edf58d45fa811cfc9da2de430374307556675c2d1c408230f994","observation_id":"c2634ffa-1e9e-4648-83f9-efd63555e76d","resolution":{"observed_at":"2026-08-10T10:34:04.553697Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.10475","last_updated":"2024-06-15T02:40:49Z","snapshot_observed_at":"2026-08-11T11:30:06.037989Z","submitted_at":"2024-06-15T02:40:49Z","title":"Discrete Latent Perspective Learning for Segmentation and Detection","version":1},"cited_work":{"arxiv_id":"2406.10475","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.10475","snapshot_observed_at":"2026-08-10T10:34:01.780403Z","title":"Discrete Latent Perspective Learning for Segmentation and Detection","venue":"cs.CV","work_id":"f9e2481b-6fad-40cc-aae7-083b41d46fa9","year":2024},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.724221Z"},"links":{"cited_paper":"/paper/2406.10475","citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:ebb6d3d9f235fc0766a81e2884e05b1168f08ec8dd7899f6e0a93c4628ea987e","observation_id":"d36ecef9-e7c6-440e-935a-58f65b41a0f7","resolution":{"observed_at":"2026-08-10T10:34:01.796964Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:04.511406Z","title":"Fast decoding in sequence models using discrete latent variables","venue":null,"work_id":"4e37dafc-23bf-42cb-baba-7d142433b7e3","year":2018},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.741056Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:a8583040aba1577d41e36b76444e7cf7591fe0dec2e5f195fd4f831a78e5a0d1","observation_id":"7b913b73-bda9-4bde-957e-29ba3d671c27","resolution":{"observed_at":"2026-08-10T10:34:04.522693Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:04.475462Z","title":"Spvit: Enabling faster vision transformers via latency-aware soft token pruning","venue":null,"work_id":"31fdc4da-1aa7-4cf0-8a0d-dd502268caee","year":2022},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.754533Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:055fcb8bef51d84d1dc5790bc22737172e23850a43b74fd0e85653381ccc06f1","observation_id":"c4540df0-bd28-42ee-8f57-48ffd3778bba","resolution":{"observed_at":"2026-08-10T10:34:04.485282Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:04.438002Z","title":"Global-local temporal representations for video person re-identification","venue":null,"work_id":"c57f3c1d-5617-4a92-b445-651e38355d6b","year":2019},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.766983Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:df1573796309c68cd5fff7cc43963683ce7e5b8177a241cc145ba2f195a1522f","observation_id":"2bd61ef7-1de5-4404-b5a0-54fe6bd6d05b","resolution":{"observed_at":"2026-08-10T10:34:04.447056Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:04.412624Z","title":"Multi-scale 3d convolu- tion network for video based person re-identification","venue":null,"work_id":"5e674aed-777a-4109-85a0-8be6ee47b6ed","year":2019},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.783289Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:20377ddfb3575ef10c659ff572eed0f4f32f199d5f13376887f46be4d3fcc1fd","observation_id":"6afe1780-49fc-4568-ba2c-0b287a391b1b","resolution":{"observed_at":"2026-08-10T10:34:04.419340Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:04.366353Z","title":"Diverse part discovery: Occluded person re-identification with part-aware transformer","venue":null,"work_id":"b9f36e8e-69cd-485a-8a6a-dab9eb1588c5","year":2021},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.791397Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:ba0360e7c9df0340822330cd561e3f51da9282b23552d9de9af24503dd21ab28","observation_id":"81f863f8-b2ba-4b09-9d35-36c03e7cb53a","resolution":{"observed_at":"2026-08-10T10:34:04.374263Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:04.333551Z","title":"Svitt: Tem- poral learning of sparse video-text transformers","venue":null,"work_id":"d1b35a34-22a8-4e45-b37c-5c567d9fe35e","year":2023},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.800153Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:e48541d1bb4f47a3eb4d3bae1d1a865048d39d10f22967f9de40ec93d4bc5c96","observation_id":"8d2f5ad6-4692-4481-94fe-254745f1180a","resolution":{"observed_at":"2026-08-10T10:34:04.340068Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:04.296327Z","title":"Efficientformer: Vision transformers at mobilenet speed","venue":null,"work_id":"4284bf86-a6b5-4b38-90e3-1970ea59f13b","year":2020},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.807634Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:daef7114317bdc1c721a1e54f5558b473f920f4b46ec1d278a5155c645a5020e","observation_id":"58878bff-9f9c-4afe-a621-d8b63b922a9b","resolution":{"observed_at":"2026-08-10T10:34:04.308115Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:04.260748Z","title":"Evit: Expediting vision transformers via token reor- ganizations","venue":null,"work_id":"5dbf9215-fa01-4f32-879a-7fe32f4422b2","year":2021},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.820143Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:4fe1485a0a7e1a30cf69d2f54db69572da92db78d975c41126c6901ba5664300","observation_id":"849ba5bd-9fbd-4a2a-a3fd-c1585051c39a","resolution":{"observed_at":"2026-08-10T10:34:04.274361Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:00.836138Z","title":"Not all patches are what you need: Expediting vision transformers via token reorganizations","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.836138Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:ddc25e704ccddf0799f01df0e841c16e829eef357974ebd07d87614674fc7974","observation_id":"3c36b440-baa5-418a-a0da-b9b18c4916ea","resolution":{"observed_at":"2026-08-10T10:34:00.836138Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:04.197666Z","title":"Supervised masked knowledge distillation for few-shot transformers","venue":null,"work_id":"38f83e9b-b583-40d9-83c4-22fd5d696ef3","year":2023},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.845032Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:578f7e5d092355798f634ca9edcf44380526f0f4d0885e1a1228c0f2da384ce4","observation_id":"cde5d3dd-fed4-471b-87f7-cccb5775bfdd","resolution":{"observed_at":"2026-08-10T10:34:04.208933Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:04.167422Z","title":"A versatile model for packet loss visibility and its application to packet prioritization","venue":null,"work_id":"c1e5fd6f-9e7f-4b7e-b65e-9a952b4c718d","year":2009},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.855441Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:79c900e359f56a5d068ec370daee742fa540f23c559da71b1164b73bde208e17","observation_id":"d0bd5b39-b890-4dba-9249-d399d71bbefd","resolution":{"observed_at":"2026-08-10T10:34:04.177601Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:04.113355Z","title":"Learning modal-invariant and temporal-memory for video-based visible-infrared person re- identification","venue":null,"work_id":"73ed3958-95d6-42cd-9dc2-bd7760e9e1d9","year":2022},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.865209Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:588671b00ce5e038126def58724b2f999f2965d8fa8d05d0158b5af3a35e904d","observation_id":"9f6afe2c-12cc-44d1-a3c9-2ed433e9d0a5","resolution":{"observed_at":"2026-08-10T10:34:04.122754Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:04.073830Z","title":"Video-based person re-identification with accumulative motion context","venue":null,"work_id":"ac1e1196-e279-4911-a730-f081c8e4dcdf","year":2017},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.875964Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:77a49e3c74d6270b2b26dbafac4fb1f268e4973a3422fdac0170740e412ef38c","observation_id":"4f6bf0da-4e50-448a-a1de-2e66d2fb3113","resolution":{"observed_at":"2026-08-10T10:34:04.088878Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:04.020657Z","title":"Spatial-temporal correlation and topology learning for person re- identification in videos","venue":null,"work_id":"a9014c4f-35ac-4af6-82fa-6f23947b5362","year":2021},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.882999Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:0368b7d0c032638dd4e5b24998c6142b323932aff513fd49eb3dac07897d4fda","observation_id":"f4d99539-83c9-4f2b-8c1f-82c38fbe9e05","resolution":{"observed_at":"2026-08-10T10:34:04.034501Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:03.992537Z","title":"Fre- quency information disentanglement network for video-based person re-identification","venue":null,"work_id":"90d88f02-d728-4e3f-9ee9-5942b4424280","year":2023},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.888267Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:6f01874142b15fb067d1c0e446a8ed7a3574f9459ef6669e0fe327e66e43b54c","observation_id":"c6fa7295-253d-468f-a89e-dd9337dd5827","resolution":{"observed_at":"2026-08-10T10:34:03.999890Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:03.966219Z","title":"Deeply coupled convolution–transformer with spatial–temporal complementary learning for video-based person re-identification","venue":null,"work_id":"19b0d321-00d9-421b-9c03-8be680c30a61","year":2023},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.897975Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:92fb3dc046d8a9dc205a42c653466e43f74c40d867aa2eb7b0894e7bb7079330","observation_id":"54402849-bfa5-4d3f-952c-485f8bc285b6","resolution":{"observed_at":"2026-08-10T10:34:03.974301Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:03.927826Z","title":"Watching you: Global-guided reciprocal learning for video-based person re-identification","venue":null,"work_id":"98b5c1a7-a951-40a4-9a93-fa387320b0ce","year":2021},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.915330Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:ac3dcd885c2b9c27e91f4a687e5803f46741ff60ba2fbd740481ce71fcc63430","observation_id":"abdb1110-f433-42b1-b98b-5c1a6d9e8df6","resolution":{"observed_at":"2026-08-10T10:34:03.940849Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:03.888646Z","title":"Noisyquant: Noisy bias-enhanced post-training activation quantization for vision transformers","venue":null,"work_id":"8caafef7-078c-419d-84c9-ac0fbb84832a","year":2023},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.924429Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:87a8d7a79905df77da964965ad14a98081a3939e8ca5c2ec05c9c2bb9638b8f3","observation_id":"7b1846f5-bab4-4e15-91f6-5f4efc4a705c","resolution":{"observed_at":"2026-08-10T10:34:03.899867Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:03.858693Z","title":"Post-training quantization for vision transformer","venue":null,"work_id":"da8a687e-e97b-4409-b7c9-12c9c85f8e35","year":2021},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.930507Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:f1ded742f817fe9ba777482cb54fc4f73bad44cbbb2abd564ff4ed29c3f94cda","observation_id":"1d5f7440-c864-4760-9294-77c39c6b51a7","resolution":{"observed_at":"2026-08-10T10:34:03.870990Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:03.827445Z","title":"Label-guided attention distillation for lane segmentation","venue":null,"work_id":"02ed70ad-bf11-4634-be83-eb9ae06fed15","year":2021},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.943975Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:852c077c6d89ca607a86676fc7368d0693ea0e02d490bee849904d620eec025f","observation_id":"ffc1e182-1a4c-48d5-a5ad-84c3e9cf45cc","resolution":{"observed_at":"2026-08-10T10:34:03.833979Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:03.802592Z","title":"Learning based multi-modality image and video compression","venue":null,"work_id":"1ac0d74a-5155-4f32-bf12-5808f0dbb1b1","year":2022},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.953962Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:f80ff674e71306edae58e80d3c7c1788fbbcc996c5b16a2152d6224e4d941148","observation_id":"008f223d-c2e5-40c2-b586-8e9b171ae2b3","resolution":{"observed_at":"2026-08-10T10:34:03.809597Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:03.774513Z","title":"Ppt: token- pruned pose transformer for monocular and multi-view human pose estimation","venue":null,"work_id":"0b810bac-d2fc-4f94-bcc4-ed077d683e38","year":null},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.961135Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:0a9bd98324efbc6d523d9322d3503f4a95f1f6f0799299c79ede4ddb60315d95","observation_id":"df999c87-0a0f-47b2-af23-0d45ca647b49","resolution":{"observed_at":"2026-08-10T10:34:03.782259Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:03.747293Z","title":"Re- current convolutional network for video-based person re-identification","venue":null,"work_id":"67d104ef-d23b-4096-91b8-36c08f123ab4","year":2016},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.966551Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:8ae1309249f6763cd5572dd1c7099b961f57fd43a7ddaff4aa1f28670863989a","observation_id":"9772c7e4-0af2-4a67-916a-7fee1d309a52","resolution":{"observed_at":"2026-08-10T10:34:03.759689Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:03.700893Z","title":"Deep spectral methods: A surprisingly strong baseline for unsupervised semantic segmentation and localization","venue":null,"work_id":"d1a6421d-c97c-4c42-a7cc-c9de2877549d","year":2022},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.971980Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:b27c238a4b3485889b55bcdc2db000df7a51af596dfdd20b22ff0bb7806cec65","observation_id":"1d7d963a-c84f-4d36-a925-bf5f54620d50","resolution":{"observed_at":"2026-08-10T10:34:03.714484Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:03.662736Z","title":"Adavit: Adaptive vision transformers for efficient image recognition","venue":null,"work_id":"cdf40492-a0ec-4089-9497-f442ae62ef39","year":2022},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.979085Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:3c84fc36b2393e2eeac60abd7eceba82be683b8dc255fb12e0328a6e06b6851c","observation_id":"02aaef4e-ed0d-4ca7-9789-230b272472b6","resolution":{"observed_at":"2026-08-10T10:34:03.670264Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:03.634974Z","title":"Counterfac- tual attention learning for fine-grained visual categorization and re- identification","venue":null,"work_id":"15146324-03c6-4d50-89a5-ca15d8fe6e98","year":2021},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.987111Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:3efba5b3315bf71a292036866d64138478841b9712bc4d69297b4caf9830d825","observation_id":"047631c8-c271-44d9-913c-f170ded2615e","resolution":{"observed_at":"2026-08-10T10:34:03.641634Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:03.595563Z","title":"Dynamicvit: Efficient vision transformers with dynamic token sparsification","venue":null,"work_id":"053a36e4-62dc-45e4-bb43-705fd82d0d54","year":2021},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:00.995065Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:366dcc21fadb4802d02d213cc969be455dd1ea1501942760c9d53bd5472e99b1","observation_id":"4aa80c06-eaaf-4974-81f7-cd986963a15e","resolution":{"observed_at":"2026-08-10T10:34:03.601500Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00714","last_updated":"2024-10-28T16:37:57Z","snapshot_observed_at":"2026-07-06T18:55:41.459417Z","submitted_at":"2024-08-01T17:00:08Z","title":"SAM 2: Segment Anything in Images and Videos","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.00714","snapshot_observed_at":"2026-08-10T10:34:01.004449Z","title":"Sam 2: Segment anything in images and videos","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:01.004449Z"},"links":{"cited_paper":"/paper/2408.00714","citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:88f137af594f0115e48d145cade616780c8bece24c25e5388da94dae0c3ccea9","observation_id":"4b95de68-333b-43fd-a312-13f738dbcf05","resolution":{"observed_at":"2026-08-10T10:34:01.004449Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:03.543120Z","title":"Co- segmentation inspired attention networks for video-based person re- identification","venue":null,"work_id":"4ea45702-6f34-4464-91be-7acaa9a8220c","year":2019},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:01.020476Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:444a39d955d3095974511ceada69ce07a2a77ac4e1b65e897a2549daef42c801","observation_id":"2e020797-d634-47ca-aded-044083e250ea","resolution":{"observed_at":"2026-08-10T10:34:03.553221Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:03.510623Z","title":"Patch slimming for efficient vision transformers","venue":null,"work_id":"66db04a7-0869-4541-9015-b05b7f2417fa","year":2022},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:01.029785Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:94237f8228643c9fb42994a6c7f00b414119c9531fa494e9be44b4b42489f730","observation_id":"94818aaa-7c94-457d-bed4-b8360c2c15e8","resolution":{"observed_at":"2026-08-10T10:34:03.522141Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:03.475276Z","title":"Multi-stage spatio-temporal aggregation transformer for video person re-identification","venue":null,"work_id":"3074f8a9-cd13-46ae-9040-4733fad54d9e","year":2022},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:01.045746Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:4cb28e3a13f8da736770bfff42b4412146c03c94196bcfc07b00630a8260d881","observation_id":"ebb60d0e-9153-4c83-b7f6-e9a0d53dc7c0","resolution":{"observed_at":"2026-08-10T10:34:03.481702Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:03.439079Z","title":"Training data-efficient image transformers & distillation through attention","venue":null,"work_id":"e68735a6-3f1f-4b18-bdf8-fdfa20638bb6","year":2021},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:01.053739Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:69d99d02e4c0def4bb36918355c4dbf30e21a3aef00a3fd474236c8736903df6","observation_id":"921d8ada-28a8-41f3-a01e-09366a84df14","resolution":{"observed_at":"2026-08-10T10:34:03.446338Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:03.391834Z","title":"Efficient video transformers with spatial-temporal token selection","venue":null,"work_id":"3c3bfbf3-d82b-441a-bd7e-1e8c6ef2dc58","year":2022},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:01.070129Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:5767d11435fd87e6195355dcf06d5ffa1bfbdaf8419745839490f0fd02482033","observation_id":"4f2d0bb4-95a6-41d1-9d37-f8429ae60f87","resolution":{"observed_at":"2026-08-10T10:34:03.405956Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:03.360780Z","title":"Pyramid spatial-temporal aggregation for video-based person re-identification","venue":null,"work_id":"cb37c079-7857-464d-a1ca-6ad1c0d823a0","year":2021},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:01.082826Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:0e8ee611ef92c00cef74e55e857cb54bcb6ce2979bb4224ad7510a0539ae0b2f","observation_id":"a7e11d25-784c-45bc-90d0-8ae283a164d5","resolution":{"observed_at":"2026-08-10T10:34:03.369877Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:03.324904Z","title":"Joint token pruning and squeezing towards more aggressive compression of vision transformers","venue":null,"work_id":"0073e872-569d-4c8c-bb58-336fc40d97e9","year":2023},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:01.095693Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:57174d495a322e4e6643d42a8ce8dbe1245955d990256b8630111e699d91073f","observation_id":"307c01be-5d1a-4f31-aeca-b5319d7704a0","resolution":{"observed_at":"2026-08-10T10:34:03.332193Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:03.276717Z","title":"Overview of the h","venue":null,"work_id":"bdda0bb8-5f35-4c2d-9191-a74ba090bc73","year":2003},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:01.115936Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:00d1dd23b8567067e1d30bef2e97c8779b5c4830f3f36a0a4fe9bb83029fa2e7","observation_id":"0aa13f07-9948-4b24-873c-767070e4e605","resolution":{"observed_at":"2026-08-10T10:34:03.299580Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:03.252862Z","title":"Cavit: Contextual alignment vision transformer for video object re-identification","venue":null,"work_id":"16a0191c-53df-446c-b720-213bde007de8","year":2022},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:01.128704Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:a908bd9006d135a082140d61b6460b8dd2aebb2beac2eac6d475e150297d1979","observation_id":"12292abd-d6f3-456e-a744-a2af9f2b9a37","resolution":{"observed_at":"2026-08-10T10:34:03.261838Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:03.225280Z","title":"Tinyvit: Fast pretraining distillation for small vision transformers","venue":null,"work_id":"1d7a891b-8abd-4e00-af97-d50236a995ce","year":2022},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:01.140111Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:882a7cb63e06b2709f3c7ff4b0dbd4147c8d53f950285a9b6561856f1dc866e6","observation_id":"cf4bce51-30f0-43f1-af45-aab88f3b625f","resolution":{"observed_at":"2026-08-10T10:34:03.236096Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:03.179220Z","title":"Learning resolution- adaptive representations for cross-resolution person re-identification","venue":null,"work_id":"0bdbe757-fcb0-48b7-b7da-818330303adf","year":2023},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:01.149464Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:92277d8401f653108d0209a7a56fc0d500069bb4fd2eb79ba2204d6e63170677","observation_id":"8a4f1536-7211-4faa-b76d-70eb0606caf7","resolution":{"observed_at":"2026-08-10T10:34:03.186474Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:03.128805Z","title":"Temporal complementarity-guided reinforcement learning for image- to-video person re-identification","venue":null,"work_id":"ff0e4d6c-8e59-4dc8-81af-374013222c48","year":2022},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:01.155537Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:7c20321830004b16b07e5531e4a5feb04626ff1546843b88604601362ccd509f","observation_id":"3db7aa8e-8600-4560-85b4-a18c9a2ae359","resolution":{"observed_at":"2026-08-10T10:34:03.140192Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:03.103239Z","title":"Adaptive graph representation learning for video person re- identification","venue":null,"work_id":"c65e9e8e-f70b-4987-9376-7c5a607d82ec","year":2020},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:01.171234Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:cf553abd243ae2b9777562733fb1e157a3d4b2cffc24f2984eab77a5e4eca285","observation_id":"31e0a4cd-b324-453f-81e6-ab3037f547d7","resolution":{"observed_at":"2026-08-10T10:34:03.110319Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:03.075570Z","title":"Segformer: Simple and efficient design for semantic segmentation with transformers","venue":null,"work_id":"f6c0bbb3-a1d3-4640-a1a2-a80bab04ffff","year":2021},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:01.180921Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:86985bbfac8c131967da44c6c2baf5bea550671fdcc634a3aab6a2b6332a09b9","observation_id":"894f8484-1cd6-4cb9-bf6c-b0f81ca7ca5f","resolution":{"observed_at":"2026-08-10T10:34:03.084236Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:03.042756Z","title":"Learning multi-granular hypergraphs for video-based person re- identification","venue":null,"work_id":"da09255d-e244-4741-9aa8-7e7dae937af1","year":2020},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:01.191992Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:53ffce66088c306ab2b682d17ffbb9c6905de2fba6581d54a0dd7c77873e6f61","observation_id":"e897bd96-5ac3-438c-9781-cdc94afa0989","resolution":{"observed_at":"2026-08-10T10:34:03.049824Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:03.011300Z","title":"Spatial-temporal graph convolutional network for video-based person re-identification","venue":null,"work_id":"7b8afdf3-5362-46de-8fb4-dffa2811bc42","year":2020},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:01.223155Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:1db8ff567def76cc13e0563fb5e42c44b25e4e973d58dae6ddef3ed4dd2b5865","observation_id":"aa82aabd-5183-453a-ac1d-84549556343e","resolution":{"observed_at":"2026-08-10T10:34:03.023721Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:02.980968Z","title":"Stfe: A comprehensive video-based person re-identification network based on spatio-temporal feature enhancement","venue":null,"work_id":"245f6f6d-7890-48af-9d80-e6ae791c783e","year":2024},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:01.239166Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:c8e87fceb26dd6a59410b7e054157d0442ccfeedd9581acb8cb38006694a6135","observation_id":"d1300d57-c14c-4878-b45b-e29a0df9bdcb","resolution":{"observed_at":"2026-08-10T10:34:02.990324Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:02.932779Z","title":"Shiftaddvit: Mixture of multiplication primitives towards efficient vision trans- former","venue":null,"work_id":"03d16786-a25b-40e2-97bb-9271e6976c71","year":2024},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:01.253538Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:00cb5e697048d3ae6a14e6fabd1c3ba51b8c5491843c7b6f9155a3e033d5b253","observation_id":"a8e9a7b2-4c4b-41b0-9b1c-cd466559acd9","resolution":{"observed_at":"2026-08-10T10:34:02.950869Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:02.899763Z","title":null,"venue":null,"work_id":"737e5eea-6cfd-414b-8577-d233bd9dc3e9","year":2020},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:01.273595Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:05114773574334cebcf0f63d704d78a91539b96e7d8e44b7f281e43f5a8fa699","observation_id":"d64b79cf-b33f-4b05-9b3b-d6d84fbf11f1","resolution":{"observed_at":"2026-08-10T10:34:02.912112Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:02.858738Z","title":"Tf-clip: Learning text-free clip for video-based person re-identification","venue":null,"work_id":"463eacac-bb1c-4f63-a091-8de6c3e9daa9","year":2024},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:01.289444Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:0049ef5ddd2efc9484003901fa22c0b3eb3f783f228898a707dfeca63e8ad74e","observation_id":"18378f3d-7930-4126-9b1c-acfc7e872773","resolution":{"observed_at":"2026-08-10T10:34:02.868769Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:02.820792Z","title":"Metaformer is actually what you need for vision","venue":null,"work_id":"f46346c8-0330-469f-b5f7-7e6fad2b425d","year":2022},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:01.303231Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:4c8e001faf82c32231ada77ecc51696a2e12cc059cb8aa86eea4e4e38fd42ad4","observation_id":"98e448b3-cf8c-443d-9d51-808a34e17e79","resolution":{"observed_at":"2026-08-10T10:34:02.829017Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:02.796255Z","title":"Ptq4vit: Post-training quantization for vision transformers with twin uniform quantization","venue":null,"work_id":"f809c0d0-f2d0-444a-896c-a25c2f9aa61b","year":2022},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:01.320177Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:e145a4be0e7bf445110f5b926a621accc65157fc315881062d52ecea99b0db10","observation_id":"d9437572-3996-44cf-9677-5dc8e707cb0d","resolution":{"observed_at":"2026-08-10T10:34:02.805325Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:02.760458Z","title":"Resmatch: Referring expression segmentation in a semi-supervised manner","venue":null,"work_id":"9fa1e54f-9db3-4c95-95c6-01919e28dddd","year":2025},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:01.328611Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:2e059b4c9e4ebb30dc8979d3e2a3f83940580afc6a6c9b2d442d8d15a1d6326b","observation_id":"903b4ee1-7d31-409a-85e6-e0761d354f86","resolution":{"observed_at":"2026-08-10T10:34:02.766980Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:02.712143Z","title":"Minivit: Compressing vision transformers with weight multiplexing","venue":null,"work_id":"3353194f-604d-4081-ac0e-fcaba549df89","year":2022},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:01.337366Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:56a9f07939f92702cd7a6dc7ddc76d9b399f25ce4cdb260e10d0461c7dc8c79f","observation_id":"c6465497-0326-48a4-9472-6a3c67ae4bd7","resolution":{"observed_at":"2026-08-10T10:34:02.721102Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:02.487322Z","title":"Magic tokens: Select diverse tokens for multi-modal object re-identification","venue":null,"work_id":"56f09d90-ef45-457c-a4d9-3f1040d17648","year":2024},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:01.344601Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:7302c8c36dca85ee70e3956993e71cc4f83ab7f712687da1377d0575671f6192","observation_id":"cf8422f9-5cbc-401c-8289-d1aae970818c","resolution":{"observed_at":"2026-08-10T10:34:02.494523Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:02.463386Z","title":"Learning bidirectional temporal cues for video-based person re-identification","venue":null,"work_id":"8222df20-092b-43ea-9c6f-49c8ed3553ab","year":2017},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:01.350272Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:7ce646b0ff48e35e501340ac652f1eaec67ceb0a64ebddc663a2eae3ac08ee4a","observation_id":"06b90b88-2f2d-4c88-a479-36a41fc82c4d","resolution":{"observed_at":"2026-08-10T10:34:02.469527Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:02.433817Z","title":"Multi- granularity reference-aided attentive feature aggregation for video- based person re-identification","venue":null,"work_id":"4b99d827-6e1f-41c5-8bd1-87c5580400ae","year":2020},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:01.358747Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:7212e0139f6d7d87f3617b861adb205d13b33e4be95704fdba5e76dce37bd5e2","observation_id":"8bbb94ba-6850-468c-9bb4-4e9b1cef9310","resolution":{"observed_at":"2026-08-10T10:34:02.444522Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:02.404407Z","title":"Structure- aware cross-modal transformer for depth completion","venue":null,"work_id":"df61682a-d5a1-4d27-960a-c31aa6f8727f","year":2024},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:01.367458Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:904f85cd033dc8d99580d3e96d136be2a528894c31666224e284f56b85827107","observation_id":"0c313ad0-a03e-4e60-b764-b4dc5a424a5b","resolution":{"observed_at":"2026-08-10T10:34:02.413916Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:02.374863Z","title":"Attribute-driven feature disentangling and temporal aggregation for video person re-identification","venue":null,"work_id":"935afc02-4395-4c72-9c85-c94eb70e3066","year":2019},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:01.374584Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:39f01398261a3c3ea859d078c0d4e471da92d7c2b0a8544f544de12fb03cfc4e","observation_id":"9b79195b-35f0-48b5-bf5f-b905e06fcb14","resolution":{"observed_at":"2026-08-10T10:34:02.385764Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:02.338735Z","title":"3d human pose estimation with spatial and temporal transformers","venue":null,"work_id":"a866517e-b19b-4d7e-8edf-e9bd1c8a05d1","year":2021},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":90,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:01.383392Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:7c4c569ab32a8642f536a2b8e5617183b3c7deb48cbc07368c511fcbbfef36d3","observation_id":"d2aa7ecb-ecf0-4824-a9b9-ebbd4bf1d86f","resolution":{"observed_at":"2026-08-10T10:34:02.351170Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1610.02984","last_updated":"2016-10-10T16:19:21Z","snapshot_observed_at":"2026-07-06T05:14:02.745611Z","submitted_at":"2016-10-10T16:19:21Z","title":"Person Re-identification: Past, Present and Future","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1610.02984","snapshot_observed_at":"2026-08-10T10:34:01.388890Z","title":"Per- son re-identification: Past, present and future","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":91,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:01.388890Z"},"links":{"cited_paper":"/paper/1610.02984","citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:c0ecb30384cb1ed2783f68fd627271b90622949f816d9fbeb04572aee4fcd8f3","observation_id":"48e554cd-45a1-4044-9b37-dad578f6e048","resolution":{"observed_at":"2026-08-10T10:34:01.388890Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:02.308865Z","title":"Joint discriminative and generative learning for person re-identification","venue":null,"work_id":"11c9efe3-1faa-4142-9ef2-7997c9f35115","year":2019},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:01.420122Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:4a9a6a72b94cd5afaa9abc02c676d2a81fdef565e299d0eece4f4cda0d867283","observation_id":"95d447e4-cb39-44ef-afde-87efb668f158","resolution":{"observed_at":"2026-08-10T10:34:02.316212Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:02.266795Z","title":"Omni-scale feature learning for person re-identification","venue":null,"work_id":"10cad0c2-cb7c-4f81-a222-ab2a1ea4a9df","year":2019},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":93,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:01.429972Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:a05a5565bd08507f2f72067de6e499e32114c6b170f983ca966655b46a635c60","observation_id":"eda5782f-e048-46b1-94a8-a659335c6634","resolution":{"observed_at":"2026-08-10T10:34:02.274860Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:02.234609Z","title":"See the forest for the trees: Joint spatial and temporal recurrent neural networks for video-based person re-identification","venue":null,"work_id":"04fb45ec-b13b-4043-861b-27ea4ce2ebcc","year":2017},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:01.440553Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:d9a0bef2e71b50e51c20b370c1a973bebabbe262a6f04b531a419af31df57323","observation_id":"c593dd8b-0ae2-4291-a6e3-eb601c39fd4b","resolution":{"observed_at":"2026-08-10T10:34:02.248839Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:02.195388Z","title":"Llafs: When large language models meet few-shot segmentation","venue":null,"work_id":"36f0748e-4f29-4065-934f-53251586087c","year":2024},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":95,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:01.449686Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:3f77d1bace759befc27f627a98c5273f48c66e21a7dc588291f01b681f8e737e","observation_id":"97d4ea06-622f-4d46-99b0-c209172392c3","resolution":{"observed_at":"2026-08-10T10:34:02.207072Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:02.166969Z","title":"Continual semantic segmentation with automatic memory sample selection","venue":null,"work_id":"d617178d-3395-4cc5-a5a5-2d297d19b99b","year":2023},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":96,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:01.457857Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:05bd9fe3a1e297a0dbae5d16e2d87dc33eb9e7a846f6f3be5ed5252ce3714b94","observation_id":"84fc00dd-55fa-4b52-9401-0faccc751e8b","resolution":{"observed_at":"2026-08-10T10:34:02.175854Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:02.140455Z","title":"Learning gabor texture features for fine-grained recognition","venue":null,"work_id":"38689cb7-e6e1-44db-b22f-00d53aba616c","year":2023},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":97,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:01.469095Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:2282db16c3e0e3f242b8a4801ee53621425e2b51d4d40cc545cf5ff800a3075f","observation_id":"9f95fbcf-1407-4e71-bec4-57dbd7de63f7","resolution":{"observed_at":"2026-08-10T10:34:02.146109Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:02.107974Z","title":"Addressing background context bias in few-shot segmentation through iterative modulation","venue":null,"work_id":"53fc2195-1a97-4983-9e8b-f941d4b56634","year":2024},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":98,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:01.484070Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:60b3a1abc2091d2331a264eb4a843baf2239061e218568c3f6efe76ddf23fd26","observation_id":"cc7719c1-30c2-4586-8b78-e6b0371cee33","resolution":{"observed_at":"2026-08-10T10:34:02.117096Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.18476","last_updated":"2024-02-28T16:57:22Z","snapshot_observed_at":"2026-08-10T13:18:36.086914Z","submitted_at":"2024-02-28T16:57:22Z","title":"IBD: Alleviating Hallucinations in Large Vision-Language Models via Image-Biased Decoding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.18476","snapshot_observed_at":"2026-08-10T10:34:01.502054Z","title":"Ibd: Alleviating hallucinations in large vision-language models via image-biased decoding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":99,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:01.502054Z"},"links":{"cited_paper":"/paper/2402.18476","citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:b0efb54f665b3ca763476b22821eb3c5b206b9aa9dd82a3fa40c64a6ad1cd24c","observation_id":"25b2325f-84a0-44fc-a614-f97f6590f3d8","resolution":{"observed_at":"2026-08-10T10:34:01.502054Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:34:02.074874Z","title":"Learning statistical texture for semantic segmentation","venue":null,"work_id":"ff538deb-3dea-44b0-a321-62ea9cb7080e","year":2021},"citing_paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification","version":1},"reference_index":100,"source":"pdf_text","source_observed_at":"2026-08-10T10:34:01.517256Z"},"links":{"citing_paper":"/paper/2501.16811"},"observation_digest":"sha256:7b79687b8107378d011e0df0856f54729116b5b961fc7e4f3a29f5a2ef72d373","observation_id":"0f656cd4-37b5-44fa-884d-ddfe7b1a491b","resolution":{"observed_at":"2026-08-10T10:34:02.082473Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2501.16811","last_updated":"2025-01-28T09:29:13Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-10T13:19:15.015567Z","submitted_at":"2025-01-28T09:29:13Z","title":"Not Every Patch is Needed: Towards a More Efficient and Effective Backbone for Video-based Person Re-identification"},"reference_resolution":{"displayed":100,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":31,"verified_exact":2,"verified_fuzzy":67},"total_outbound_references":102},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"thesis":"As of 11 August 2026, this Paper Citation Record lists 100 of 102 outbound references and 1 inbound Pith citation observation for arXiv:2501.16811."}