{"as_of":"2026-08-10T19:35:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:c2df360a33e0ab266d0bdb8454fbb7bfc4d39289dac3c6c8dcf3be62bc0d0295","coverage":[{"denominator":47,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":47,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-01T07:07:44.009137Z","state":"measured"},{"denominator":47,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":47,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2607.21576/citation-record","integrity":"/paper/2607.21576/integrity","json":"/paper/2607.21576/citation-record.json","paper":"/paper/2607.21576"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-01T07:07:42.700451Z","title":"V-jepa 2: Self-supervised video models enable understanding, prediction and planning.arXiv preprint arXiv:2506.09985,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.21576","last_updated":"2026-07-23T17:55:07Z","snapshot_observed_at":"2026-08-08T01:47:00.528938Z","submitted_at":"2026-07-23T17:55:07Z","title":"Self-Supervised Learning of Structured Dynamics from Videos","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-01T07:07:42.700451Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2607.21576"},"observation_digest":"sha256:277879558cb086df998a8df56402a708f9808d7c00dba78757ef99403b024ba8","observation_id":"fcf5aec1-d427-4934-92b1-88dbb6c38a08","resolution":{"observed_at":"2026-08-01T07:07:42.700451Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.19468","last_updated":"2025-07-25T17:54:10Z","snapshot_observed_at":"2026-08-08T15:35:43.235215Z","submitted_at":"2025-07-25T17:54:10Z","title":"Back to the Features: DINO as a Foundation for Video World Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.19468","snapshot_observed_at":"2026-08-01T07:07:42.787898Z","title":"Back to the features: Dino as a foundation for video world models.arXiv preprint arXiv:2507.19468, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.21576","last_updated":"2026-07-23T17:55:07Z","snapshot_observed_at":"2026-08-08T01:47:00.528938Z","submitted_at":"2026-07-23T17:55:07Z","title":"Self-Supervised Learning of Structured Dynamics from Videos","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-01T07:07:42.787898Z"},"links":{"cited_paper":"/paper/2507.19468","citing_paper":"/paper/2607.21576"},"observation_digest":"sha256:6c4f911878178ea6a48b3ea1b536d50cf36ef446ae258de981fd0d3d11dd2054","observation_id":"645df39c-c401-4de0-81ef-3b53daab1ac7","resolution":{"observed_at":"2026-08-01T07:07:42.787898Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T07:07:42.875142Z","title":"Revisiting feature prediction for learning visual representations from video.Transactions on Machine Learning Research, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.21576","last_updated":"2026-07-23T17:55:07Z","snapshot_observed_at":"2026-08-08T01:47:00.528938Z","submitted_at":"2026-07-23T17:55:07Z","title":"Self-Supervised Learning of Structured Dynamics from Videos","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-01T07:07:42.875142Z"},"links":{"citing_paper":"/paper/2607.21576"},"observation_digest":"sha256:8eb51afdc7c1b374f77cddedd09fa187481c0721ca5467cc356b509eafa6a89d","observation_id":"e57671f3-34fe-4506-9642-2ad1495ef298","resolution":{"observed_at":"2026-08-01T07:07:42.875142Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T07:07:42.965780Z","title":"VFMF: World modeling by forecasting vision foundation model features.arXiv preprint arXiv:2512.11225,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.21576","last_updated":"2026-07-23T17:55:07Z","snapshot_observed_at":"2026-08-08T01:47:00.528938Z","submitted_at":"2026-07-23T17:55:07Z","title":"Self-Supervised Learning of Structured Dynamics from Videos","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-01T07:07:42.965780Z"},"links":{"citing_paper":"/paper/2607.21576"},"observation_digest":"sha256:435f0ec6d1014aeb0008e8017afe3d1261ed70893c7cd4c65e051873a507c316","observation_id":"ff4ba9dd-6950-4f73-9f34-ac9508a804ce","resolution":{"observed_at":"2026-08-01T07:07:42.965780Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2604.12012","last_updated":"2026-04-13T20:00:04Z","snapshot_observed_at":"2026-07-06T23:00:17.634307Z","submitted_at":"2026-04-13T20:00:04Z","title":"TIPSv2: Advancing Vision-Language Pretraining with Enhanced Patch-Text Alignment","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2604.12012","snapshot_observed_at":"2026-08-01T07:07:43.056117Z","title":"TIPSv2: Advancing vision- language pretraining with enhanced patch-text alignment.arXiv preprint arXiv:2604.12012,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.21576","last_updated":"2026-07-23T17:55:07Z","snapshot_observed_at":"2026-08-08T01:47:00.528938Z","submitted_at":"2026-07-23T17:55:07Z","title":"Self-Supervised Learning of Structured Dynamics from Videos","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-01T07:07:43.056117Z"},"links":{"cited_paper":"/paper/2604.12012","citing_paper":"/paper/2607.21576"},"observation_digest":"sha256:fca9cf679738bbce93984d735c2ff7cb6090ccc33076bf969b130bad3ca661be","observation_id":"de75f00c-62f0-4962-9043-dd50da99aee3","resolution":{"observed_at":"2026-08-01T07:07:43.056117Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1907.06987","last_updated":"2022-10-17T19:58:40Z","snapshot_observed_at":"2026-07-06T08:08:03.081236Z","submitted_at":"2019-07-15T12:58:21Z","title":"A Short Note on the Kinetics-700 Human Action Dataset","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1907.06987","snapshot_observed_at":"2026-08-01T07:07:43.147125Z","title":"A short note on the kinetics- 700 human action dataset.arXiv preprint arXiv:1907.06987, 2019","venue":null,"work_id":null,"year":1907},"citing_paper":{"arxiv_id":"2607.21576","last_updated":"2026-07-23T17:55:07Z","snapshot_observed_at":"2026-08-08T01:47:00.528938Z","submitted_at":"2026-07-23T17:55:07Z","title":"Self-Supervised Learning of Structured Dynamics from Videos","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-01T07:07:43.147125Z"},"links":{"cited_paper":"/paper/1907.06987","citing_paper":"/paper/2607.21576"},"observation_digest":"sha256:534eadf16262db183ee65684289ea8e0fef1c6bf16fd9b71b08b815553f11e6e","observation_id":"e902e301-623b-4c03-9de9-fc397afdd936","resolution":{"observed_at":"2026-08-01T07:07:43.147125Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15212","last_updated":"2025-07-09T16:58:07Z","snapshot_observed_at":"2026-08-06T10:25:48.518948Z","submitted_at":"2024-12-19T18:59:51Z","title":"Scaling 4D Representations","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15212","snapshot_observed_at":"2026-08-01T07:07:43.307749Z","title":"Scaling 4d representations.arXiv preprint arXiv:2412.15212, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.21576","last_updated":"2026-07-23T17:55:07Z","snapshot_observed_at":"2026-08-08T01:47:00.528938Z","submitted_at":"2026-07-23T17:55:07Z","title":"Self-Supervised Learning of Structured Dynamics from Videos","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-01T07:07:43.307749Z"},"links":{"cited_paper":"/paper/2412.15212","citing_paper":"/paper/2607.21576"},"observation_digest":"sha256:c308681383989002e674e6e176595104087d24817e542f7c2c0e4e12610bf445","observation_id":"9b3a923d-c9c3-4986-9f24-86606157a485","resolution":{"observed_at":"2026-08-01T07:07:43.307749Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T07:07:43.455509Z","title":"Vision transformers need registers","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.21576","last_updated":"2026-07-23T17:55:07Z","snapshot_observed_at":"2026-08-08T01:47:00.528938Z","submitted_at":"2026-07-23T17:55:07Z","title":"Self-Supervised Learning of Structured Dynamics from Videos","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-01T07:07:43.455509Z"},"links":{"citing_paper":"/paper/2607.21576"},"observation_digest":"sha256:7636daf26b6a15adc4e04fff758f65c4020d086e5433bc02ebd82e6d3097bc60","observation_id":"fe9778c6-5fa5-4553-aa9b-0415daf664ba","resolution":{"observed_at":"2026-08-01T07:07:43.455509Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T07:07:43.538067Z","title":"Probing the 3d awareness of visual foundation models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.21576","last_updated":"2026-07-23T17:55:07Z","snapshot_observed_at":"2026-08-08T01:47:00.528938Z","submitted_at":"2026-07-23T17:55:07Z","title":"Self-Supervised Learning of Structured Dynamics from Videos","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-01T07:07:43.538067Z"},"links":{"citing_paper":"/paper/2607.21576"},"observation_digest":"sha256:6448aa051fd9506d0825f6850495811d0cd80eceec0b0a8239d2b8975ee73a06","observation_id":"672b183b-0146-4b90-b179-08ac7841accb","resolution":{"observed_at":"2026-08-01T07:07:43.538067Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T07:07:43.612958Z","title":"Scalable pre-training of large autoregressive image models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.21576","last_updated":"2026-07-23T17:55:07Z","snapshot_observed_at":"2026-08-08T01:47:00.528938Z","submitted_at":"2026-07-23T17:55:07Z","title":"Self-Supervised Learning of Structured Dynamics from Videos","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-01T07:07:43.612958Z"},"links":{"citing_paper":"/paper/2607.21576"},"observation_digest":"sha256:4899727faa4ea67dd435139fbc4296433bfbad39a1b64ab087eb40477f1864e0","observation_id":"8c1e3832-5293-4856-a12b-ae26f0fa47d4","resolution":{"observed_at":"2026-08-01T07:07:43.612958Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T07:07:43.715885Z","title":"Multimodal autoregressive pre-training of large vision encoders","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.21576","last_updated":"2026-07-23T17:55:07Z","snapshot_observed_at":"2026-08-08T01:47:00.528938Z","submitted_at":"2026-07-23T17:55:07Z","title":"Self-Supervised Learning of Structured Dynamics from Videos","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-01T07:07:43.715885Z"},"links":{"citing_paper":"/paper/2607.21576"},"observation_digest":"sha256:b6ecedd5155a91df6040f58925705f90337056c3fa093c63a68d3176c8a2ce9e","observation_id":"4f62d957-313a-43c0-bc5f-819ab38c4242","resolution":{"observed_at":"2026-08-01T07:07:43.715885Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T07:07:43.835002Z","title":"Learning latent action world models in the wild.arXiv preprint arXiv:2601.05230,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.21576","last_updated":"2026-07-23T17:55:07Z","snapshot_observed_at":"2026-08-08T01:47:00.528938Z","submitted_at":"2026-07-23T17:55:07Z","title":"Self-Supervised Learning of Structured Dynamics from Videos","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-01T07:07:43.835002Z"},"links":{"citing_paper":"/paper/2607.21576"},"observation_digest":"sha256:78c1abeaf6c1eb9fbff5d4e97485edf8e1607c35299943b9ca0072581de0f59a","observation_id":"594d648b-9a2c-4095-a313-23e54a26b8d7","resolution":{"observed_at":"2026-08-01T07:07:43.835002Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T07:07:43.848667Z","title":"something something","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2607.21576","last_updated":"2026-07-23T17:55:07Z","snapshot_observed_at":"2026-08-08T01:47:00.528938Z","submitted_at":"2026-07-23T17:55:07Z","title":"Self-Supervised Learning of Structured Dynamics from Videos","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-01T07:07:43.848667Z"},"links":{"citing_paper":"/paper/2607.21576"},"observation_digest":"sha256:dbed1577556509b6d12c9a6359266467b2368a5f0fce0339c3442f3aeef7bf76","observation_id":"21ed614d-83ad-45e1-ad25-fb18370dd025","resolution":{"observed_at":"2026-08-01T07:07:43.848667Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T07:07:43.920666Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.21576","last_updated":"2026-07-23T17:55:07Z","snapshot_observed_at":"2026-08-08T01:47:00.528938Z","submitted_at":"2026-07-23T17:55:07Z","title":"Self-Supervised Learning of Structured Dynamics from Videos","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-01T07:07:43.920666Z"},"links":{"citing_paper":"/paper/2607.21576"},"observation_digest":"sha256:f59d9fb327d364a2efb6e795a4c2ca35168081fcf222642c69e0f30fe19c9cab","observation_id":"1a5c7dea-afcc-4dae-959e-3bfa5d651e49","resolution":{"observed_at":"2026-08-01T07:07:43.920666Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T07:07:43.934201Z","title":"Siamese masked autoencoders.Advances in Neural Information Processing Systems, 36:40676–40693, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.21576","last_updated":"2026-07-23T17:55:07Z","snapshot_observed_at":"2026-08-08T01:47:00.528938Z","submitted_at":"2026-07-23T17:55:07Z","title":"Self-Supervised Learning of Structured Dynamics from Videos","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-01T07:07:43.934201Z"},"links":{"citing_paper":"/paper/2607.21576"},"observation_digest":"sha256:cc0fba75fb3fa834f9e98df111e2d05de5c7bbff7ead6ed782b6ffc288a569d4","observation_id":"f873e337-e9d3-47b8-bf4b-d10e18b844d4","resolution":{"observed_at":"2026-08-01T07:07:43.934201Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T07:07:43.936695Z","title":"Masked autoencoders are scalable vision learners","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.21576","last_updated":"2026-07-23T17:55:07Z","snapshot_observed_at":"2026-08-08T01:47:00.528938Z","submitted_at":"2026-07-23T17:55:07Z","title":"Self-Supervised Learning of Structured Dynamics from Videos","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-01T07:07:43.936695Z"},"links":{"citing_paper":"/paper/2607.21576"},"observation_digest":"sha256:d4b64df98a9bb1e3f9100f1cd4c509ebfeb8edce8e3681027cba40ad8b22c7c8","observation_id":"547a2166-ea33-4e0a-bb48-c3eee0d8b9d3","resolution":{"observed_at":"2026-08-01T07:07:43.936695Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T07:07:43.938975Z","title":"Rotary position embedding for vision transformer","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.21576","last_updated":"2026-07-23T17:55:07Z","snapshot_observed_at":"2026-08-08T01:47:00.528938Z","submitted_at":"2026-07-23T17:55:07Z","title":"Self-Supervised Learning of Structured Dynamics from Videos","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-01T07:07:43.938975Z"},"links":{"citing_paper":"/paper/2607.21576"},"observation_digest":"sha256:287c85334d029094faae2b4c0f4c20b6b86d02176c26fe930a2dca63c3e7073f","observation_id":"22fc5f9e-72c4-4c09-b583-80c4ac259aef","resolution":{"observed_at":"2026-08-01T07:07:43.938975Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T07:07:43.941401Z","title":"VGGT4D: Mining motion cues in visual geometry transformers for 4d scene reconstruction.arXiv preprint arXiv:2511.19971, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.21576","last_updated":"2026-07-23T17:55:07Z","snapshot_observed_at":"2026-08-08T01:47:00.528938Z","submitted_at":"2026-07-23T17:55:07Z","title":"Self-Supervised Learning of Structured Dynamics from Videos","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-01T07:07:43.941401Z"},"links":{"citing_paper":"/paper/2607.21576"},"observation_digest":"sha256:89d82535092201b177e79f3f06a9aa2ea8cd12b7320804ad5379eec65ca95126","observation_id":"c1d0eed5-b78e-4a04-97f1-8097b2ed3d86","resolution":{"observed_at":"2026-08-01T07:07:43.941401Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2604.04913","last_updated":"2026-04-06T17:55:05Z","snapshot_observed_at":"2026-08-10T11:36:36.197748Z","submitted_at":"2026-04-06T17:55:05Z","title":"A Frame is Worth One Token: Efficient Generative World Modeling with Delta Tokens","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2604.04913","snapshot_observed_at":"2026-08-01T07:07:43.943838Z","title":"A frame is worth one token: Efficient generative world modeling with delta tokens.arXiv preprint arXiv:2604.04913, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.21576","last_updated":"2026-07-23T17:55:07Z","snapshot_observed_at":"2026-08-08T01:47:00.528938Z","submitted_at":"2026-07-23T17:55:07Z","title":"Self-Supervised Learning of Structured Dynamics from Videos","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-01T07:07:43.943838Z"},"links":{"cited_paper":"/paper/2604.04913","citing_paper":"/paper/2607.21576"},"observation_digest":"sha256:562cfa69abf838c533d9dcc73ea2691f41cf1aeb717d9c904ebd782e20cd3ee2","observation_id":"a8425ea1-c34c-4e13-bdcd-5392d11716e8","resolution":{"observed_at":"2026-08-01T07:07:43.943838Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2511.10647","last_updated":"2025-11-13T18:59:53Z","snapshot_observed_at":"2026-07-06T22:35:46.018050Z","submitted_at":"2025-11-13T18:59:53Z","title":"Depth Anything 3: Recovering the Visual Space from Any Views","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2511.10647","snapshot_observed_at":"2026-08-01T07:07:43.946535Z","title":"Depth anything 3: Recovering the visual space from any views.arXiv preprint arXiv:2511.10647, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.21576","last_updated":"2026-07-23T17:55:07Z","snapshot_observed_at":"2026-08-08T01:47:00.528938Z","submitted_at":"2026-07-23T17:55:07Z","title":"Self-Supervised Learning of Structured Dynamics from Videos","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-01T07:07:43.946535Z"},"links":{"cited_paper":"/paper/2511.10647","citing_paper":"/paper/2607.21576"},"observation_digest":"sha256:8a9116071159fa746edce87d3cd5c457a014e97ee3c8a4be02a917e5331c0bad","observation_id":"b3c5fe62-51b3-41a6-b595-f5cc7558a7ac","resolution":{"observed_at":"2026-08-01T07:07:43.946535Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.15376","last_updated":"2025-08-29T09:26:15Z","snapshot_observed_at":"2026-08-07T16:00:29.105170Z","submitted_at":"2025-04-21T18:34:57Z","title":"Towards Understanding Camera Motions in Any Video","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.15376","snapshot_observed_at":"2026-08-01T07:07:43.948987Z","title":"Towards understanding camera motions in any video.arXiv preprint arXiv:2504.15376, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.21576","last_updated":"2026-07-23T17:55:07Z","snapshot_observed_at":"2026-08-08T01:47:00.528938Z","submitted_at":"2026-07-23T17:55:07Z","title":"Self-Supervised Learning of Structured Dynamics from Videos","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-01T07:07:43.948987Z"},"links":{"cited_paper":"/paper/2504.15376","citing_paper":"/paper/2607.21576"},"observation_digest":"sha256:eb7b22f0636f3cdd9c00bda91c5978ca09aef21d4e5055fb28bd70b70d16024a","observation_id":"28d0b606-de80-41da-bdf2-c07efc56d5bc","resolution":{"observed_at":"2026-08-01T07:07:43.948987Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T07:07:43.951383Z","title":"DL3DV-10k: A large-scale scene dataset for deep learning-based 3d vision","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.21576","last_updated":"2026-07-23T17:55:07Z","snapshot_observed_at":"2026-08-08T01:47:00.528938Z","submitted_at":"2026-07-23T17:55:07Z","title":"Self-Supervised Learning of Structured Dynamics from Videos","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-01T07:07:43.951383Z"},"links":{"citing_paper":"/paper/2607.21576"},"observation_digest":"sha256:1654d04054736909ea18115a20f12c8c114ebee78dbe2eda2b2ccc1de7378aff","observation_id":"f03464ab-f014-4252-9354-1a64f1527cb4","resolution":{"observed_at":"2026-08-01T07:07:43.951383Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2602.10094","last_updated":"2026-05-05T16:49:24Z","snapshot_observed_at":"2026-07-06T22:45:20.365624Z","submitted_at":"2026-02-10T18:57:04Z","title":"4RC: 4D Reconstruction via Conditional Querying Anytime and Anywhere","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2602.10094","snapshot_observed_at":"2026-08-01T07:07:43.953662Z","title":"4RC: 4d recon- struction via conditional querying anytime and anywhere.arXiv preprint arXiv:2602.10094,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.21576","last_updated":"2026-07-23T17:55:07Z","snapshot_observed_at":"2026-08-08T01:47:00.528938Z","submitted_at":"2026-07-23T17:55:07Z","title":"Self-Supervised Learning of Structured Dynamics from Videos","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-01T07:07:43.953662Z"},"links":{"cited_paper":"/paper/2602.10094","citing_paper":"/paper/2607.21576"},"observation_digest":"sha256:46d60f49813193fd710b42915f5d3c9e3e6b120b81a7b2ad3ad84c9dc8dcb79e","observation_id":"43ae3062-fec4-4b39-99d4-c39099bfd6d5","resolution":{"observed_at":"2026-08-01T07:07:43.953662Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2603.14482","last_updated":"2026-06-11T10:07:56Z","snapshot_observed_at":"2026-08-06T07:23:27.418432Z","submitted_at":"2026-03-15T17:02:40Z","title":"V-JEPA 2.1: Unlocking Dense Features in Video Self-Supervised Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2603.14482","snapshot_observed_at":"2026-08-01T07:07:43.956666Z","title":"V-JEPA 2.1: Unlocking dense features in video self-supervised learning.arXiv preprint arXiv:2603.14482, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.21576","last_updated":"2026-07-23T17:55:07Z","snapshot_observed_at":"2026-08-08T01:47:00.528938Z","submitted_at":"2026-07-23T17:55:07Z","title":"Self-Supervised Learning of Structured Dynamics from Videos","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-01T07:07:43.956666Z"},"links":{"cited_paper":"/paper/2603.14482","citing_paper":"/paper/2607.21576"},"observation_digest":"sha256:7603aec6311f45ca6d7445062809c83f213b20dc7985eecc88c8a13ddbcfd5e0","observation_id":"7a833b1c-186f-4aab-974b-2a97f257b8d4","resolution":{"observed_at":"2026-08-01T07:07:43.956666Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T07:07:43.958940Z","title":"DINOv2: Learning robust visual features without supervision.Transactions on Machine Learning Research, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.21576","last_updated":"2026-07-23T17:55:07Z","snapshot_observed_at":"2026-08-08T01:47:00.528938Z","submitted_at":"2026-07-23T17:55:07Z","title":"Self-Supervised Learning of Structured Dynamics from Videos","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-01T07:07:43.958940Z"},"links":{"citing_paper":"/paper/2607.21576"},"observation_digest":"sha256:a4ba45eb092db36902d3016e1d57a3c4257886c82a66190b9a96e9acd38ed9d9","observation_id":"34d14f7c-97fd-43c8-ac36-5edf4ac34380","resolution":{"observed_at":"2026-08-01T07:07:43.958940Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1704.00675","last_updated":"2018-03-01T17:50:08Z","snapshot_observed_at":"2026-08-02T10:51:13.194643Z","submitted_at":"2017-04-03T16:44:46Z","title":"The 2017 DAVIS Challenge on Video Object Segmentation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1704.00675","snapshot_observed_at":"2026-08-01T07:07:43.961142Z","title":"The 2017 davis challenge on video object segmentation.arXiv:1704.00675,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2607.21576","last_updated":"2026-07-23T17:55:07Z","snapshot_observed_at":"2026-08-08T01:47:00.528938Z","submitted_at":"2026-07-23T17:55:07Z","title":"Self-Supervised Learning of Structured Dynamics from Videos","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-01T07:07:43.961142Z"},"links":{"cited_paper":"/paper/1704.00675","citing_paper":"/paper/2607.21576"},"observation_digest":"sha256:46e8a6616c76b6f96c2b0490ee671fe774b1d4ff2bcf74ade61a995a0b9c82a5","observation_id":"182abe4a-1f14-4b48-82db-e75561513901","resolution":{"observed_at":"2026-08-01T07:07:43.961142Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T07:07:43.963461Z","title":"Time does tell: Self-supervised time-tuning of dense image representations","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.21576","last_updated":"2026-07-23T17:55:07Z","snapshot_observed_at":"2026-08-08T01:47:00.528938Z","submitted_at":"2026-07-23T17:55:07Z","title":"Self-Supervised Learning of Structured Dynamics from Videos","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-01T07:07:43.963461Z"},"links":{"citing_paper":"/paper/2607.21576"},"observation_digest":"sha256:5c80287f064fc292ac3cc03393dbd9e87568780fca746ffe59328f5d973c7b8b","observation_id":"4583a188-4775-43e9-a355-cbe4adf0db60","resolution":{"observed_at":"2026-08-01T07:07:43.963461Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T07:07:43.965731Z","title":"MoSiC: Optimal-transport motion trajectory for dense self- supervised learning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.21576","last_updated":"2026-07-23T17:55:07Z","snapshot_observed_at":"2026-08-08T01:47:00.528938Z","submitted_at":"2026-07-23T17:55:07Z","title":"Self-Supervised Learning of Structured Dynamics from Videos","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-01T07:07:43.965731Z"},"links":{"citing_paper":"/paper/2607.21576"},"observation_digest":"sha256:3c3eab535413ff487de7f17d6959a9aeb0896a42140a583c4813b868f53e58e8","observation_id":"450b75fa-98fa-4a39-89df-87ec8183382e","resolution":{"observed_at":"2026-08-01T07:07:43.965731Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.10104","last_updated":"2025-08-13T18:00:55Z","snapshot_observed_at":"2026-07-06T22:12:35.584339Z","submitted_at":"2025-08-13T18:00:55Z","title":"DINOv3","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.10104","snapshot_observed_at":"2026-08-01T07:07:43.967955Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.21576","last_updated":"2026-07-23T17:55:07Z","snapshot_observed_at":"2026-08-08T01:47:00.528938Z","submitted_at":"2026-07-23T17:55:07Z","title":"Self-Supervised Learning of Structured Dynamics from Videos","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-01T07:07:43.967955Z"},"links":{"cited_paper":"/paper/2508.10104","citing_paper":"/paper/2607.21576"},"observation_digest":"sha256:60a8313fb114274943cc18c7ede63e8d9c3e4a964e6babe14c3e84a60dc270aa","observation_id":"fa113fe4-b7ac-4da2-bf39-3462434d4f0a","resolution":{"observed_at":"2026-08-01T07:07:43.967955Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T07:07:43.970153Z","title":"Roformer: Enhanced transformer with rotary position embedding.Neurocomputing, 568:127063, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.21576","last_updated":"2026-07-23T17:55:07Z","snapshot_observed_at":"2026-08-08T01:47:00.528938Z","submitted_at":"2026-07-23T17:55:07Z","title":"Self-Supervised Learning of Structured Dynamics from Videos","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-01T07:07:43.970153Z"},"links":{"citing_paper":"/paper/2607.21576"},"observation_digest":"sha256:86cdffe0dc43564a3d4c604f65524453d1a82f17e101badb6ff457661afb4e0e","observation_id":"8ea85e0e-05bb-4463-92be-8ee667b13647","resolution":{"observed_at":"2026-08-01T07:07:43.970153Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T07:07:43.972365Z","title":"VideoMAE: Masked autoencoders are data-efficient learners for self-supervised video pre-training.Advances in neural information processing systems, 35:10078–10093, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.21576","last_updated":"2026-07-23T17:55:07Z","snapshot_observed_at":"2026-08-08T01:47:00.528938Z","submitted_at":"2026-07-23T17:55:07Z","title":"Self-Supervised Learning of Structured Dynamics from Videos","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-01T07:07:43.972365Z"},"links":{"citing_paper":"/paper/2607.21576"},"observation_digest":"sha256:351962c6464ab1b08a8a9bc027ab12f0c25ae93e1eae8c7cbf5d70eb80c29d8f","observation_id":"2645d77d-e259-461a-a17a-7ac3cebee984","resolution":{"observed_at":"2026-08-01T07:07:43.972365Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14786","last_updated":"2025-02-20T18:08:29Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-20T18:08:29Z","title":"SigLIP 2: Multilingual Vision-Language Encoders with Improved Semantic Understanding, Localization, and Dense Features","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14786","snapshot_observed_at":"2026-08-01T07:07:43.974627Z","title":"SigLIP 2: Multilingual vision-language encoders with improved semantic understanding, local- ization, and dense features.arXiv preprint arXiv:2502.14786, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.21576","last_updated":"2026-07-23T17:55:07Z","snapshot_observed_at":"2026-08-08T01:47:00.528938Z","submitted_at":"2026-07-23T17:55:07Z","title":"Self-Supervised Learning of Structured Dynamics from Videos","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-01T07:07:43.974627Z"},"links":{"cited_paper":"/paper/2502.14786","citing_paper":"/paper/2607.21576"},"observation_digest":"sha256:59b84102f309a8ff2ad781e5f07be1181445b1700d75ea1f2d52e1fc78116bdb","observation_id":"2a851265-df0a-4b5a-9b3a-ae8248a7c565","resolution":{"observed_at":"2026-08-01T07:07:43.974627Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T07:07:43.977072Z","title":"Is imagenet worth 1 video? learning strong image encoders from 1 long unlabelled video","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.21576","last_updated":"2026-07-23T17:55:07Z","snapshot_observed_at":"2026-08-08T01:47:00.528938Z","submitted_at":"2026-07-23T17:55:07Z","title":"Self-Supervised Learning of Structured Dynamics from Videos","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-01T07:07:43.977072Z"},"links":{"citing_paper":"/paper/2607.21576"},"observation_digest":"sha256:652b2927df2bea213571777cbacbddc950954cc755b8cfcae54934beef35466b","observation_id":"5aa556d7-3cee-41b0-8ff0-63f37f49b2c4","resolution":{"observed_at":"2026-08-01T07:07:43.977072Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T07:07:43.979519Z","title":"PooDLe: Pooled and dense self-supervised learning from naturalistic videos","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.21576","last_updated":"2026-07-23T17:55:07Z","snapshot_observed_at":"2026-08-08T01:47:00.528938Z","submitted_at":"2026-07-23T17:55:07Z","title":"Self-Supervised Learning of Structured Dynamics from Videos","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-01T07:07:43.979519Z"},"links":{"citing_paper":"/paper/2607.21576"},"observation_digest":"sha256:d3e96e12f908334d06e7d26309e510c6e5882faac812101f476435d6d6fae2cc","observation_id":"99103198-8a40-45ba-9adf-1fe3682195b8","resolution":{"observed_at":"2026-08-01T07:07:43.979519Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T07:07:43.981930Z","title":"VGGT: Visual geometry grounded transformer","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.21576","last_updated":"2026-07-23T17:55:07Z","snapshot_observed_at":"2026-08-08T01:47:00.528938Z","submitted_at":"2026-07-23T17:55:07Z","title":"Self-Supervised Learning of Structured Dynamics from Videos","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-01T07:07:43.981930Z"},"links":{"citing_paper":"/paper/2607.21576"},"observation_digest":"sha256:a1c604868c6263717caee043aafa22ec389205e607c9bb88aa24546dccb9a645","observation_id":"709e7f6f-3816-4c21-9a39-cb847bbe7d88","resolution":{"observed_at":"2026-08-01T07:07:43.981930Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T07:07:43.984503Z","title":"VideoMAE v2: Scaling video masked autoencoders with dual masking","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.21576","last_updated":"2026-07-23T17:55:07Z","snapshot_observed_at":"2026-08-08T01:47:00.528938Z","submitted_at":"2026-07-23T17:55:07Z","title":"Self-Supervised Learning of Structured Dynamics from Videos","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-01T07:07:43.984503Z"},"links":{"citing_paper":"/paper/2607.21576"},"observation_digest":"sha256:7cf91ed3df6c08c42578319bdc35354ecb3046d466804b080459ea9d92df3001","observation_id":"c973aec4-f6bb-4ab7-8b38-b14ba09dcfab","resolution":{"observed_at":"2026-08-01T07:07:43.984503Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T07:07:43.986832Z","title":"Continuous 3d perception model with persistent state","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.21576","last_updated":"2026-07-23T17:55:07Z","snapshot_observed_at":"2026-08-08T01:47:00.528938Z","submitted_at":"2026-07-23T17:55:07Z","title":"Self-Supervised Learning of Structured Dynamics from Videos","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-01T07:07:43.986832Z"},"links":{"citing_paper":"/paper/2607.21576"},"observation_digest":"sha256:58ad8ebba5c497ccb528f38671f71e5f5ff787a63b0bd94c9c5440dba8748b42","observation_id":"89c50610-fdc6-4295-9893-a0331c937ea6","resolution":{"observed_at":"2026-08-01T07:07:43.986832Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T07:07:43.989146Z","title":"DUSt3R: Geometric 3d vision made easy","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.21576","last_updated":"2026-07-23T17:55:07Z","snapshot_observed_at":"2026-08-08T01:47:00.528938Z","submitted_at":"2026-07-23T17:55:07Z","title":"Self-Supervised Learning of Structured Dynamics from Videos","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-01T07:07:43.989146Z"},"links":{"citing_paper":"/paper/2607.21576"},"observation_digest":"sha256:dd02d625b6df89cd226cb8c54dc563fae4f2ccc8830238968bc6f303b8019023","observation_id":"58664195-27a9-44fd-b628-84dbfd619214","resolution":{"observed_at":"2026-08-01T07:07:43.989146Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.13347","last_updated":"2026-03-07T07:01:59Z","snapshot_observed_at":"2026-08-02T12:12:15.465550Z","submitted_at":"2025-07-17T17:59:53Z","title":"$\\pi^3$: Permutation-Equivariant Visual Geometry Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.13347","snapshot_observed_at":"2026-08-01T07:07:43.991386Z","title":"pi3: Permutation-equivariant visual geometry learning.arXiv preprint arXiv:2507.13347, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.21576","last_updated":"2026-07-23T17:55:07Z","snapshot_observed_at":"2026-08-08T01:47:00.528938Z","submitted_at":"2026-07-23T17:55:07Z","title":"Self-Supervised Learning of Structured Dynamics from Videos","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-01T07:07:43.991386Z"},"links":{"cited_paper":"/paper/2507.13347","citing_paper":"/paper/2607.21576"},"observation_digest":"sha256:547286d7a4ca8159da87cb47c28516a9d4f5b587da4e716f8d6c1ef393cafd79","observation_id":"bd6afa90-1371-427b-b1c3-e89398834c6e","resolution":{"observed_at":"2026-08-01T07:07:43.991386Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T07:07:43.993570Z","title":"CroCo: Self-supervised pre-training for 3d vision tasks by cross-view completion.Advances in Neural Information Processing Systems, 35:3502–3516, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.21576","last_updated":"2026-07-23T17:55:07Z","snapshot_observed_at":"2026-08-08T01:47:00.528938Z","submitted_at":"2026-07-23T17:55:07Z","title":"Self-Supervised Learning of Structured Dynamics from Videos","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-01T07:07:43.993570Z"},"links":{"citing_paper":"/paper/2607.21576"},"observation_digest":"sha256:358df2919685bd0178c3fb988bcb5ca5cdebb76e6c2521c2ac18ca1f7a78e759","observation_id":"fde9c463-4331-40e4-ab44-ae8814e8eb1e","resolution":{"observed_at":"2026-08-01T07:07:43.993570Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T07:07:43.995703Z","title":"YouTube-VOS: Sequence-to-sequence video object segmentation","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2607.21576","last_updated":"2026-07-23T17:55:07Z","snapshot_observed_at":"2026-08-08T01:47:00.528938Z","submitted_at":"2026-07-23T17:55:07Z","title":"Self-Supervised Learning of Structured Dynamics from Videos","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-01T07:07:43.995703Z"},"links":{"citing_paper":"/paper/2607.21576"},"observation_digest":"sha256:d3be367e567f2e0b16bc367170be6c4b493abbe94b2d6a044540b820597b7b2d","observation_id":"892cc710-3da3-4e36-81e7-21a4b860634c","resolution":{"observed_at":"2026-08-01T07:07:43.995703Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1809.03327","last_updated":"2018-09-06T04:19:45Z","snapshot_observed_at":"2026-07-06T07:00:12.122241Z","submitted_at":"2018-09-06T04:19:45Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1809.03327","snapshot_observed_at":"2026-08-01T07:07:43.997932Z","title":"YouTube-VOS: A large-scale video object segmentation benchmark.arXiv preprint arXiv:1809.03327, 2018","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2607.21576","last_updated":"2026-07-23T17:55:07Z","snapshot_observed_at":"2026-08-08T01:47:00.528938Z","submitted_at":"2026-07-23T17:55:07Z","title":"Self-Supervised Learning of Structured Dynamics from Videos","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-01T07:07:43.997932Z"},"links":{"cited_paper":"/paper/1809.03327","citing_paper":"/paper/2607.21576"},"observation_digest":"sha256:eb1a38ab8de3c58b2be46dd2088996f78ccb0ea0b9dc229dd50d017f0ecdee24","observation_id":"35b55820-2081-4f19-8147-b4dd6e62b70f","resolution":{"observed_at":"2026-08-01T07:07:43.997932Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T07:07:44.000290Z","title":"In pursuit of pixel supervision for visual pre-training.arXiv preprint arXiv:2512.15715, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.21576","last_updated":"2026-07-23T17:55:07Z","snapshot_observed_at":"2026-08-08T01:47:00.528938Z","submitted_at":"2026-07-23T17:55:07Z","title":"Self-Supervised Learning of Structured Dynamics from Videos","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-01T07:07:44.000290Z"},"links":{"citing_paper":"/paper/2607.21576"},"observation_digest":"sha256:3ff913e91046b445c971ce98e240806baac6f0fd3236d85a989e85b0ccaf7010","observation_id":"c8f125c4-0e5e-4172-b6aa-e76d34d9bae5","resolution":{"observed_at":"2026-08-01T07:07:44.000290Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T07:07:44.002629Z","title":"Sigmoid loss for language image pre-training","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.21576","last_updated":"2026-07-23T17:55:07Z","snapshot_observed_at":"2026-08-08T01:47:00.528938Z","submitted_at":"2026-07-23T17:55:07Z","title":"Self-Supervised Learning of Structured Dynamics from Videos","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-01T07:07:44.002629Z"},"links":{"citing_paper":"/paper/2607.21576"},"observation_digest":"sha256:c55392c9578d46179d7320699734178cd422ab3d96c5d767dadd722afa531f9c","observation_id":"f04e3e93-a621-4f88-bbe7-a42d78715838","resolution":{"observed_at":"2026-08-01T07:07:44.002629Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T07:07:44.004881Z","title":"MonST3R: A simple approach for estimating geometry in the presence of motion","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.21576","last_updated":"2026-07-23T17:55:07Z","snapshot_observed_at":"2026-08-08T01:47:00.528938Z","submitted_at":"2026-07-23T17:55:07Z","title":"Self-Supervised Learning of Structured Dynamics from Videos","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-01T07:07:44.004881Z"},"links":{"citing_paper":"/paper/2607.21576"},"observation_digest":"sha256:f5a5055c1c61f30e911811f01041f2d08be873cf2ba9809e03ba1d35e4c23001","observation_id":"e869fc90-2e3a-43d9-89d8-815c453154a3","resolution":{"observed_at":"2026-08-01T07:07:44.004881Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T07:07:44.007097Z","title":"DINO-WM: World models on pre-trained visual features enable zero-shot planning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.21576","last_updated":"2026-07-23T17:55:07Z","snapshot_observed_at":"2026-08-08T01:47:00.528938Z","submitted_at":"2026-07-23T17:55:07Z","title":"Self-Supervised Learning of Structured Dynamics from Videos","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-01T07:07:44.007097Z"},"links":{"citing_paper":"/paper/2607.21576"},"observation_digest":"sha256:2813dfcb1f9d0f6663d3f27e3c71f5d5026b2fcf155636e4b3f69aa9888e258a","observation_id":"87e6ebbf-0bee-495f-86a2-ceff86ce66f4","resolution":{"observed_at":"2026-08-01T07:07:44.007097Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T07:07:44.009137Z","title":"Recurrent video masked autoencoders","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.21576","last_updated":"2026-07-23T17:55:07Z","snapshot_observed_at":"2026-08-08T01:47:00.528938Z","submitted_at":"2026-07-23T17:55:07Z","title":"Self-Supervised Learning of Structured Dynamics from Videos","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-01T07:07:44.009137Z"},"links":{"citing_paper":"/paper/2607.21576"},"observation_digest":"sha256:cd9c40969e527d1e678b7143ef1c46cc5fa0de381d45ff2bf052e735a5e46aa7","observation_id":"bfebb87d-d8e2-43b6-85dc-94f29411fda9","resolution":{"observed_at":"2026-08-01T07:07:44.009137Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2607.21576","last_updated":"2026-07-23T17:55:07Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-08T01:47:00.528938Z","submitted_at":"2026-07-23T17:55:07Z","title":"Self-Supervised Learning of Structured Dynamics from Videos"},"reference_resolution":{"displayed":47,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":46,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":47},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 47 of 47 outbound references and 0 inbound Pith citation observations for arXiv:2607.21576."}