{"as_of":"2026-08-10T09:35:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:5ede845c6a29754835eb872a376aecb70a0a995690a2840f2d168e6388f7e838","coverage":[{"denominator":43,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":43,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T11:55:28.601657Z","state":"measured"},{"denominator":44,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":44,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-22T05:41:39.396469Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-22T05:44:38.774124Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"cited_work":{"arxiv_id":"2506.01119","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.01119","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"MOOSE: Pay atten- tion to temporal dynamics for video understanding via optical flows.arXiv preprint arXiv:2506.01119","venue":null,"work_id":"73be5edd-6d6a-4339-bd54-db823a83eb7d","year":null},"citing_paper":{"arxiv_id":"2605.22823","last_updated":"2026-05-21T17:59:56Z","snapshot_observed_at":"2026-07-30T22:55:34.671958Z","submitted_at":"2026-05-21T17:59:56Z","title":"Which Way Did It Move? Diagnosing and Overcoming Directional Motion Blindness in Video-LLMs","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-22T05:41:39.396469Z"},"links":{"cited_paper":"/paper/2506.01119","citing_paper":"/paper/2605.22823"},"observation_digest":"sha256:e7d41d1bf45df1a1a4f60b985b0a508bb413fc7dfa3d6cfe603b628af7627764","observation_id":"3d7e4a8f-ac48-45cc-8d2e-b9a52f16edca","resolution":{"observed_at":"2026-05-22T05:44:38.777481Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2506.01119/citation-record","integrity":"/paper/2506.01119/integrity","json":"/paper/2506.01119/citation-record.json","paper":"/paper/2506.01119"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.986741Z","title":"Gundavarapu, Liangzhe Yuan, Hao Zhou, Shen Yan, Jennifer J","venue":null,"work_id":"9a7410f2-ba18-4f49-bcdb-58cc6c0731a6","year":2024},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.386030Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:b248460f83e245b7aaf8d612eb471860f7da97add25d48e9e35662a6942f8997","observation_id":"6dbce686-65b1-4798-8744-c1c3aa65eeae","resolution":{"observed_at":"2026-08-07T11:55:28.990505Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1705.06950","last_updated":"2017-05-19T12:07:01Z","snapshot_observed_at":"2026-08-08T17:46:50.107463Z","submitted_at":"2017-05-19T12:07:01Z","title":"The Kinetics Human Action Video Dataset","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1705.06950","snapshot_observed_at":"2026-08-07T11:55:28.393424Z","title":"The kinetics human action video dataset.ArXiv, abs/1705.06950, 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.393424Z"},"links":{"cited_paper":"/paper/1705.06950","citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:5f8cf993ccd6c7112552c4bec83488d6afd0d0d51d82fef65fbbd993721e7e01","observation_id":"31296b17-4e1d-4577-ad09-ca09a7c82ab2","resolution":{"observed_at":"2026-08-07T11:55:28.393424Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.978528Z","title":"The human visual system and its role in motion perception","venue":null,"work_id":"5cea4598-f940-457b-8299-e6ddee1e5861","year":2011},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.402383Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:dcb583c7fac66b011486c8f194b942049727270957055ddb9ee73770c2c082eb","observation_id":"b7d5e1b8-23fc-4be0-917f-49c2ba3768fc","resolution":{"observed_at":"2026-08-07T11:55:28.981697Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2102.05095","last_updated":"2021-06-09T14:48:13Z","snapshot_observed_at":"2026-08-09T19:37:59.145976Z","submitted_at":"2021-02-09T19:49:33Z","title":"Is Space-Time Attention All You Need for Video Understanding?","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2102.05095","snapshot_observed_at":"2026-08-07T11:55:28.412053Z","title":"Is space-time attention all you need for video understanding?CoRR, abs/2102.05095, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.412053Z"},"links":{"cited_paper":"/paper/2102.05095","citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:ff2b568b4c3f7398f9803cce2f69b3a0884a89b0fcbe8de02334d390640994a6","observation_id":"5ab418a0-6bfc-4f97-990e-a8b928f6f186","resolution":{"observed_at":"2026-08-07T11:55:28.412053Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.969278Z","title":"Vivit: A video vision transformer.2021 IEEE/CVF International Conference on Computer Vision (ICCV), pages 6816–6826, 2021","venue":null,"work_id":"4fba4926-ec02-4b52-a49e-048118521b1a","year":2021},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.422695Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:a4bb522770725dc727a79558e678f5a28bae32a5566af75ba0ee3e1205c7f947","observation_id":"46b2f413-8943-40e6-8cb2-27593dab6d00","resolution":{"observed_at":"2026-08-07T11:55:28.973074Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.960768Z","title":"De Gruyter, Berlin, Boston, 2016","venue":null,"work_id":"3792dcdf-79e5-4f93-bf59-9d163fb2dad4","year":2016},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.431253Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:a3af6fcc026e0b070389bef1b6ec543a859f5c9b9a53312c1fb5ae2ca787a615","observation_id":"dfb07648-2b4f-402e-9a76-45ec7b1d00b5","resolution":{"observed_at":"2026-08-07T11:55:28.963711Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.951816Z","title":"Action recognition for surveillance applications using optic flow and svm","venue":null,"work_id":"0eae86cd-8a18-49fd-ad32-eea5c36004b5","year":2007},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.438232Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:d3ad0a771043bc650cf4546cf36d588c1f6e2deb78c08d0dd1a3ced512a8e7e6","observation_id":"a7113b0f-7476-4a94-a79a-75a6b8fdb899","resolution":{"observed_at":"2026-08-07T11:55:28.955123Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.942710Z","title":"Conv3d-based video violence detection network using optical flow and rgb data.Sensors, 24(2):317, 2024","venue":null,"work_id":"4fda9b6c-6a5b-4a7c-97fa-0fc937983ec8","year":2024},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.450359Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:a54831a68fc20c9856509ba77b03bf0cc4bc531b52aa9c5f58aa6d923de35130","observation_id":"d4797cdf-ccde-4b0e-ba25-09f695a2a5c8","resolution":{"observed_at":"2026-08-07T11:55:28.946060Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.932893Z","title":"A multi-modal egocentric activity recognition approach towards video domain generalization.Sensors, 24(8):2491, 2024","venue":null,"work_id":"cba87012-5d7a-4a20-b473-0176e5ec6631","year":2024},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.458569Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:ff632769800ee8e1311dbedab094d3f781f77abd0518cf239b66e6380bcfee8c","observation_id":"247d453b-637f-4fcc-8dcb-6eb04aab0fad","resolution":{"observed_at":"2026-08-07T11:55:28.936770Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.924032Z","title":"Nayak, and Shrikanth S","venue":null,"work_id":"2e4f177f-4d5a-4835-8725-e00b155ec341","year":2021},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.465645Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:f4d2a7cee6d50754e4e65d6146ea1ea00b78ac95d8df8de5a3f0fa1851486b60","observation_id":"30e22078-01a4-4dd2-aa6a-52eb4cde4e96","resolution":{"observed_at":"2026-08-07T11:55:28.927244Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.915121Z","title":"Kosloski, Siddhi Patel, Zeke A","venue":null,"work_id":"6bf57b0e-d105-4fc1-abf3-de9831b6eec7","year":2025},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.473893Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:c91ef3f7cdca3a36068bf7bcbfe1fc0599420794d874c5d715df16c0d888b6a6","observation_id":"48e6121a-cc97-4115-b8b8-8d0048e1dc14","resolution":{"observed_at":"2026-08-07T11:55:28.918941Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.905630Z","title":"Childplay: A new benchmark for understanding children’s gaze behaviour","venue":null,"work_id":"8e681077-a1ad-4b86-9873-e864016cc1b2","year":2023},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.481578Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:e5d22c24bfcddc4f54fab09f0adee259928a3e05d710d7b73b7752467d163051","observation_id":"3f00602f-9e2d-4f81-8fbc-2185311414f8","resolution":{"observed_at":"2026-08-07T11:55:28.909200Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.896471Z","title":"Barner, and Roghayeh Leila Barmaki","venue":null,"work_id":"18c2c5f1-eac7-42f1-887f-ec4d098670d6","year":2023},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.492403Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:8f71c09a1fa46fe928437bca9ae9140fd92ecb7a7d231fae439bbeed24624fba","observation_id":"bb5d62d2-7286-4735-90b4-4d41ed1d31e2","resolution":{"observed_at":"2026-08-07T11:55:28.900092Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.887216Z","title":"Reversible vision transformers","venue":null,"work_id":"1fa13a9a-c0aa-4883-b230-832580461d18","year":2022},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.501974Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:96eea773c93bc91a3160c9f511ca04df38fd4b3176d9e42b452b99776f18e7ab","observation_id":"dcc67fb1-afe9-4b82-9f4c-4b6c84bf1209","resolution":{"observed_at":"2026-08-07T11:55:28.890570Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.878278Z","title":"Multiscale vision transformers","venue":null,"work_id":"7814bb41-e189-49be-9b01-ed7cc0ab8636","year":2021},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.505343Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:5e3ae8460487d08405d3cc33203d74439590648578b9a1f2f7fa659e779b6bf5","observation_id":"94e91b00-0004-4897-9bb8-5b2b282ea082","resolution":{"observed_at":"2026-08-07T11:55:28.881706Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.869490Z","title":"X3d: Expanding architectures for efficient video recognition","venue":null,"work_id":"502ff843-c569-4859-9143-bd2c984a514f","year":2020},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.508483Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:456961424d0c5c8e4201c0e2634d89adb68aecce99d6a6b5dfc6dfa43d240470","observation_id":"fe739488-63d3-493d-9d5e-3d0e52ce9576","resolution":{"observed_at":"2026-08-07T11:55:28.872697Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.860690Z","title":"A large-scale study on unsupervised spatiotemporal representation learning","venue":null,"work_id":"a2a4f910-13ca-480c-8e9e-3d02556c342c","year":2021},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.511429Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:63e399a272d53db77bbe9227af6fb8c8e0bc18000ce95f02d0c8da5a941a70c7","observation_id":"6d16135b-8ed2-40d3-8f03-1628219ff299","resolution":{"observed_at":"2026-08-07T11:55:28.863939Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.851335Z","title":"Spatio- temporal collaborative module for efficient action recognition.IEEE Transactions on Image Processing, 31:7279–7291, 2022","venue":null,"work_id":"f9a0208a-2258-4a5d-a7cb-44405a0656f3","year":2022},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.514882Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:222675a9d32d29b804674dc6b4ad2a3d1d5dac0e14b8d6de08079e37b1015b1b","observation_id":"06197e95-cd25-434e-b15c-0c73abe5192e","resolution":{"observed_at":"2026-08-07T11:55:28.854792Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.841988Z","title":"Quo vadis, action recognition? a new model and the kinetics dataset","venue":null,"work_id":"89901d67-0ad5-491c-8c9a-63e46bd9bf40","year":2017},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.518323Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:6ab3a46f1fe7f1a0624e44ec3ce871ab4a444fd7cc6a10278f2d718cf2f3b6c2","observation_id":"ea230c41-9731-4304-9776-7504f3072ba9","resolution":{"observed_at":"2026-08-07T11:55:28.845441Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.832997Z","title":"Batch transformer: Look for attention in batch, 2024","venue":null,"work_id":"1ef233a5-5eb8-43cd-a424-28231e157c26","year":2024},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.521658Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:2a8f0a71693eea8065e42bf027706926dedd2f48a3df48639f1daf1b20b9ba19","observation_id":"469ec2df-3213-41ce-8db6-06f55bcefd40","resolution":{"observed_at":"2026-08-07T11:55:28.836220Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.823990Z","title":"Videomae: masked autoencoders are data-efficient learners for self-supervised video pre-training","venue":null,"work_id":"5e0c3110-037d-43d7-952c-dec33b1f3b52","year":2022},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.525238Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:557cd6947549549981b3324a1e13f6b9edb1fcebaf749f65319429beed7e7e12","observation_id":"3bd54dc5-556c-4af1-bb20-de4ee4845e82","resolution":{"observed_at":"2026-08-07T11:55:28.827404Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.815145Z","title":"Videomae v2: Scaling video masked autoencoders with dual masking","venue":null,"work_id":"779620ec-f7ff-4a4e-9fbb-4d3191b0a81b","year":2023},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.528471Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:4d3b38af7136311313e1764b22fbea9a2aec12024c3eeef66c7de478e55d5a93","observation_id":"2867165f-1f18-46b0-b603-81f640b3bc1d","resolution":{"observed_at":"2026-08-07T11:55:28.818394Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.805770Z","title":"Video swin transformer","venue":null,"work_id":"4b294883-c7db-4a55-8b48-47993c2dd580","year":2022},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.531356Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:ddb602d8c06099a920a72a944aedc29d787160092b180a0b2c34c97d920e4f5a","observation_id":"01f74cea-7896-458c-a9d7-63a76d406320","resolution":{"observed_at":"2026-08-07T11:55:28.809366Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.534382Z","title":"Pyslowfast","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.534382Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:e88e5f6cf98761cd518e755f892c70d84d5c8fdb16eb27ecb9ae9999d6fa2a63","observation_id":"a4733bde-7530-4e67-893c-184e944c8ecb","resolution":{"observed_at":"2026-08-07T11:55:28.534382Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.791312Z","title":"Jampani, Andreas Geiger, and Michael J","venue":null,"work_id":"85a33d1a-2b84-4678-b43d-da201c45c8d8","year":2017},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.537466Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:d8c733803e6f401bcf1eb622ba8c97c5ec5c640dc0acf358310d715b5b6428d1","observation_id":"b58e5f88-f92c-4727-b35e-a2b9a5c8914b","resolution":{"observed_at":"2026-08-07T11:55:28.794641Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.781746Z","title":"Henriques","venue":null,"work_id":"0d615f8e-4ee2-4d9d-add1-fd93a20cfdad","year":2021},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.540883Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:a3a010b776ddd5c49f763ed89644c362b8b219302d5c788bddffaae4d33c5e00","observation_id":"d35cb0e2-e7ea-4b65-bac0-d7ecd28ba1b0","resolution":{"observed_at":"2026-08-07T11:55:28.784880Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.772936Z","title":"Memflow: Optical flow estimation and prediction with memory, 2024","venue":null,"work_id":"9b4c5983-d639-4869-979a-5ffa70f8026e","year":2024},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.543940Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:1c8c4ded1ec29ba3c34742e3c057ddd3200923a2cf841fa7f358a2816165e122","observation_id":"f6fd2541-73a6-4144-957a-d50c8ae420b2","resolution":{"observed_at":"2026-08-07T11:55:28.776070Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2001.04451","last_updated":"2020-02-18T16:01:18Z","snapshot_observed_at":"2026-07-06T08:50:12.690900Z","submitted_at":"2020-01-13T18:38:28Z","title":"Reformer: The Efficient Transformer","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2001.04451","snapshot_observed_at":"2026-08-07T11:55:28.547233Z","title":"Reformer: The efficient transformer","venue":null,"work_id":null,"year":2001},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.547233Z"},"links":{"cited_paper":"/paper/2001.04451","citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:96aca2be583e4dc5df65004b813f8527715decfe2f7b96e36351055b6fe146ca","observation_id":"fe05929b-d2e5-4667-8e2a-a6d2f7bc8fa8","resolution":{"observed_at":"2026-08-07T11:55:28.547233Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.00752","last_updated":"2024-05-31T17:55:27Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-01T18:01:34Z","title":"Mamba: Linear-Time Sequence Modeling with Selective State Spaces","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.00752","snapshot_observed_at":"2026-08-07T11:55:28.551969Z","title":"Mamba: Linear-time sequence modeling with selective state spaces","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.551969Z"},"links":{"cited_paper":"/paper/2312.00752","citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:96ad63b5ae1efdb80676d3b4b83777adc172bf31d50f978d501c4b10b1480f6c","observation_id":"fdd2c949-005b-4f33-9cbf-a3c67a4305b0","resolution":{"observed_at":"2026-08-07T11:55:28.551969Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.763188Z","title":"Raft: Recurrent all-pairs field transforms for optical flow","venue":null,"work_id":"f797f9e2-5915-4692-aede-db48d81eaf74","year":2020},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.555565Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:d86b6c5b4f87a01fd6d2cc09bbe9f9d7c991036c45fbbf1598892659551c7b2f","observation_id":"97448051-ee0f-460c-b750-ae3501c43f0c","resolution":{"observed_at":"2026-08-07T11:55:28.766560Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.754299Z","title":null,"venue":null,"work_id":"7fa4b4f4-9a30-4368-804f-79b29416e5c5","year":2012},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.559252Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:49e85cd644d139233099c9851fbba08bc382708d3773c2dd00030752a2669bc2","observation_id":"39b7184f-ab40-4135-bc7d-b3a461216f70","resolution":{"observed_at":"2026-08-07T11:55:28.757518Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.744398Z","title":"Vision transformers need registers, 2023","venue":null,"work_id":"e56c2bef-42c0-471d-bc05-5c67680f8095","year":2023},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.562573Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:6b4326b41832b8adc8338efdab8022afc0b05568023da3a48ac591d6266bf233","observation_id":"f48e9839-fec3-489c-9e26-7f1322ae2cd6","resolution":{"observed_at":"2026-08-07T11:55:28.748304Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.565743Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.565743Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:ed76d7c22394748e04f07cb1dcfad6ca312c3e1785b9e6d0e2e836f58b471912","observation_id":"48a7837c-a8b6-4d22-8428-c966ac4a442c","resolution":{"observed_at":"2026-08-07T11:55:28.565743Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.730864Z","title":"something something","venue":null,"work_id":"c8d785ca-7080-4533-bd3d-5cbdd420cc61","year":2017},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.569311Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:1b1dc0721d9ee0c11e22055335ace8ccf180e7aca0f9f6be940eed6c6028d714","observation_id":"6f8d3a28-c38d-4d96-9cf0-917e8ebab219","resolution":{"observed_at":"2026-08-07T11:55:28.733996Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.720851Z","title":"Haa500: Human-centric atomic action dataset with curated videos.2021 IEEE/CVF International Conference on Computer Vision (ICCV), pages 13445–13454, 2020","venue":null,"work_id":"83e4f5e5-d086-4165-a211-fc848959c79b","year":2021},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.572573Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:466342ea66192ba106fad17b4c3e2eecc75f5887ad25715e43a26e84522a86d9","observation_id":"3ff158d3-b578-4830-8336-25fd38b8a102","resolution":{"observed_at":"2026-08-07T11:55:28.724868Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1212.0402","last_updated":"2012-12-03T14:45:31Z","snapshot_observed_at":"2026-07-06T03:01:10.229407Z","submitted_at":"2012-12-03T14:45:31Z","title":"UCF101: A Dataset of 101 Human Actions Classes From Videos in The Wild","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1212.0402","snapshot_observed_at":"2026-08-07T11:55:28.576141Z","title":"Ucf101: A dataset of 101 human actions classes from videos in the wild.ArXiv, abs/1212.0402, 2012","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.576141Z"},"links":{"cited_paper":"/paper/1212.0402","citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:30d138c3bf366f58e8daf8018ee3615b541ef18c7e1e252ea5162064156059f7","observation_id":"a183b072-d9e6-47f7-b98a-ecb2db274ecf","resolution":{"observed_at":"2026-08-07T11:55:28.576141Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.02858","last_updated":"2023-10-25T06:23:31Z","snapshot_observed_at":"2026-07-06T15:38:39.712379Z","submitted_at":"2023-06-05T13:17:27Z","title":"Video-LLaMA: An Instruction-tuned Audio-Visual Language Model for Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.02858","snapshot_observed_at":"2026-08-07T11:55:28.579305Z","title":"Video-llama: An instruction-tuned audio-visual language model for video understanding.arXiv preprint arXiv:2306.02858, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.579305Z"},"links":{"cited_paper":"/paper/2306.02858","citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:7c3bb878f51e66db58c2b1978d3d7f32640a0fac49f6c4aa5301646782d3f4b9","observation_id":"49b04534-6599-45ae-8e45-a02de3856930","resolution":{"observed_at":"2026-08-07T11:55:28.579305Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.07476","last_updated":"2024-10-30T06:49:54Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-11T17:22:23Z","title":"VideoLLaMA 2: Advancing Spatial-Temporal Modeling and Audio Understanding in Video-LLMs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.07476","snapshot_observed_at":"2026-08-07T11:55:28.582939Z","title":"Videollama 2: Advancing spatial- temporal modeling and audio understanding in video-llms.arXiv preprint arXiv:2406.07476, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.582939Z"},"links":{"cited_paper":"/paper/2406.07476","citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:c5c99e8620ed9e4da7f8435452a45b34d8391781bbfe0883816f569a956cd470","observation_id":"11b6460c-27c4-44d8-a83a-db3a97378f3b","resolution":{"observed_at":"2026-08-07T11:55:28.582939Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13106","last_updated":"2025-06-03T03:33:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T18:59:46Z","title":"VideoLLaMA 3: Frontier Multimodal Foundation Models for Image and Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13106","snapshot_observed_at":"2026-08-07T11:55:28.586764Z","title":"Boqiang Zhang","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.586764Z"},"links":{"cited_paper":"/paper/2501.13106","citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:680cc108ee321ec55c12c39479434047c1f90d5983962637de2cfd4b5472c88d","observation_id":"6a77fe62-b43d-4c1f-a132-461310eeec73","resolution":{"observed_at":"2026-08-07T11:55:28.586764Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15841","last_updated":"2024-09-15T05:00:18Z","snapshot_observed_at":"2026-08-08T03:13:28.000968Z","submitted_at":"2024-07-22T17:58:04Z","title":"SlowFast-LLaVA: A Strong Training-Free Baseline for Video Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15841","snapshot_observed_at":"2026-08-07T11:55:28.590775Z","title":"Xu Mingze","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.590775Z"},"links":{"cited_paper":"/paper/2407.15841","citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:1dcbd6ab8c9cae09dd99079ccdf61462b8ff59bb37d9baf17f8ef0a63a823423","observation_id":"2ca677bd-85df-4f23-bf31-c6e4b77df4bb","resolution":{"observed_at":"2026-08-07T11:55:28.590775Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.10122","last_updated":"2024-10-01T12:07:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-11-16T10:59:44Z","title":"Video-LLaVA: Learning United Visual Representation by Alignment Before Projection","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.10122","snapshot_observed_at":"2026-08-07T11:55:28.593941Z","title":"Video-llava: Learning united visual representation by alignment before projection.arXiv preprint arXiv:2311.10122, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.593941Z"},"links":{"cited_paper":"/paper/2311.10122","citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:98655be13ac8498645380d9dfe3c8b7319fe6509fdc5d4d01fddebb26eac46fd","observation_id":"38aef75a-669e-4a59-9487-ed095bf3ccb4","resolution":{"observed_at":"2026-08-07T11:55:28.593941Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01852","last_updated":"2024-01-22T03:11:15Z","snapshot_observed_at":"2026-08-07T05:10:33.059352Z","submitted_at":"2023-10-03T07:33:27Z","title":"LanguageBind: Extending Video-Language Pretraining to N-modality by Language-based Semantic Alignment","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01852","snapshot_observed_at":"2026-08-07T11:55:28.598004Z","title":"Languagebind: Extending video-language pretraining to n-modality by language-based semantic alignment.arXiv preprint arXiv:2310.01852, 2023","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.598004Z"},"links":{"cited_paper":"/paper/2310.01852","citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:b63bc982fb8a5f057f12253e96c8fe674982a9e6b1610e9028837cbc05ac4e94","observation_id":"f8d45bca-9cee-4951-a3ea-6d6c44723162","resolution":{"observed_at":"2026-08-07T11:55:28.598004Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.709577Z","title":"running” or “jumping","venue":null,"work_id":"4a617660-3970-4da8-92e4-c18f85620539","year":2024},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.601657Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:9d3de8bb86105331cbc1b12a1499c08e31bfe135b6527036a74354da7dec88b1","observation_id":"ba947274-ad79-41ca-93f7-98a7c4503750","resolution":{"observed_at":"2026-08-07T11:55:28.713614Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows"},"reference_resolution":{"displayed":43,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":14,"verified_exact":0,"verified_fuzzy":29},"total_outbound_references":43},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 43 of 43 outbound references and 1 inbound Pith citation observation for arXiv:2506.01119."}