{"as_of":"2026-08-07T07:20:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:a4b8f9276ac323f1d7ec9edd3e4493e3727d759ceebbcc47892391294f3373b5","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":35,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":35,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":35,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":35,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T05:35:50.906801Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T06:39:37.380967Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":"2109.14084","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-07-04T06:39:37.380967Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":"509d3b12-48dd-4984-9d28-a28d917f0b1c","year":2021},"citing_paper":{"arxiv_id":"2203.12601","last_updated":"2022-11-18T05:57:09Z","snapshot_observed_at":"2026-08-02T11:48:59.131827Z","submitted_at":"2022-03-23T17:55:09Z","title":"R3M: A Universal Visual Representation for Robot Manipulation","version":3},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-05-15T13:26:53.843613Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2203.12601"},"observation_digest":"sha256:33bed028ad54f00522dab4f43856299d00dd94090e967b196e77f533630900ec","observation_id":"fe34e4e4-4f16-43b4-b379-342a81ff67a2","resolution":{"observed_at":"2026-05-15T13:26:53.979248Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":"2109.14084","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-07-04T06:39:37.380967Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":"509d3b12-48dd-4984-9d28-a28d917f0b1c","year":2021},"citing_paper":{"arxiv_id":"2204.00598","last_updated":"2022-05-27T17:52:50Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-04-01T17:43:13Z","title":"Socratic Models: Composing Zero-Shot Multimodal Reasoning with Language","version":2},"reference_index":95,"source":"pdf_text","source_observed_at":"2026-05-16T09:50:00.546571Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2204.00598"},"observation_digest":"sha256:2c0d39bee4bac869b6455a9a1d33e20bc5046c47cfb955946aae946c8925f4dd","observation_id":"72761ef6-7d17-4890-8c76-6cabcc30f73d","resolution":{"observed_at":"2026-05-16T09:50:00.608579Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":"2109.14084","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-07-04T06:39:37.380967Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":"509d3b12-48dd-4984-9d28-a28d917f0b1c","year":2021},"citing_paper":{"arxiv_id":"2212.03191","last_updated":"2022-12-07T12:20:55Z","snapshot_observed_at":"2026-07-06T14:27:34.639236Z","submitted_at":"2022-12-06T18:09:49Z","title":"InternVideo: General Video Foundation Models via Generative and Discriminative Learning","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-17T00:36:53.235740Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2212.03191"},"observation_digest":"sha256:d7371c0f4178da698b3fa1f90d99e320da9e121c4f17761659fd270880124196","observation_id":"1eaf6c1f-7f3f-4dbd-8fe1-793a40eb840d","resolution":{"observed_at":"2026-05-17T00:36:53.372393Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":"2109.14084","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-07-04T06:39:37.380967Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":"509d3b12-48dd-4984-9d28-a28d917f0b1c","year":2021},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:d98229189fdd0306bb797732cb70b07837a71192161579bb5d37cd6c0173299d","observation_id":"b9b1d4cd-aa91-48eb-b583-b569e1e5952e","resolution":{"observed_at":"2026-05-13T23:30:00.675840Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":"2109.14084","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-07-04T06:39:37.380967Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":"509d3b12-48dd-4984-9d28-a28d917f0b1c","year":2021},"citing_paper":{"arxiv_id":"2307.06942","last_updated":"2024-01-04T05:00:34Z","snapshot_observed_at":"2026-07-06T15:53:46.393481Z","submitted_at":"2023-07-13T17:58:32Z","title":"InternVid: A Large-scale Video-Text Dataset for Multimodal Understanding and Generation","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-15T06:30:22.431538Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2307.06942"},"observation_digest":"sha256:5d6573cb3060ef17dbff7d612b3bfd84b7351c75de3ea29410d7de2ee11450b3","observation_id":"eed293b8-bc7f-4447-a1a7-672174934fd9","resolution":{"observed_at":"2026-05-15T06:30:22.554543Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":"2109.14084","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-07-04T06:39:37.380967Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":"509d3b12-48dd-4984-9d28-a28d917f0b1c","year":2021},"citing_paper":{"arxiv_id":"2310.01852","last_updated":"2024-01-22T03:11:15Z","snapshot_observed_at":"2026-08-07T05:10:33.059352Z","submitted_at":"2023-10-03T07:33:27Z","title":"LanguageBind: Extending Video-Language Pretraining to N-modality by Language-based Semantic Alignment","version":7},"reference_index":177,"source":"arxiv_source","source_observed_at":"2026-05-17T03:27:58.952076Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2310.01852"},"observation_digest":"sha256:e2f2f82ef25b1a2d829c7cf2d3861f82d22b6b14cbb91c97b1449ba4272395c2","observation_id":"f4791b68-3222-428b-9824-0255dc411a5c","resolution":{"observed_at":"2026-05-17T03:27:59.117146Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":"2109.14084","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-07-04T06:39:37.380967Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":"509d3b12-48dd-4984-9d28-a28d917f0b1c","year":2021},"citing_paper":{"arxiv_id":"2404.08471","last_updated":"2024-02-15T18:59:11Z","snapshot_observed_at":"2026-08-07T02:30:11.447693Z","submitted_at":"2024-02-15T18:59:11Z","title":"Revisiting Feature Prediction for Learning Visual Representations from Video","version":1},"reference_index":292,"source":"arxiv_source","source_observed_at":"2026-05-12T12:40:23.709098Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2404.08471"},"observation_digest":"sha256:743d8fc95aa6ff228b093e5da29bdae219059c1691c9af56accf70ebad99295d","observation_id":"9d4cba46-2e95-48a7-8e42-e8a936162640","resolution":{"observed_at":"2026-05-12T12:40:24.082290Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":"2109.14084","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-07-04T06:39:37.380967Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":"509d3b12-48dd-4984-9d28-a28d917f0b1c","year":2021},"citing_paper":{"arxiv_id":"2503.13821","last_updated":"2026-04-03T20:02:54Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-18T01:57:48Z","title":"Stitch-a-Demo: Video Demonstrations from Multistep Descriptions","version":3},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-05-23T00:30:55.729900Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2503.13821"},"observation_digest":"sha256:0672f1f622889d0a1ee591e3629da60fa3faeeb2a7482081f59b3e0e4752c067","observation_id":"56f2d5c3-e0e1-47d8-a3ce-3dc8cb32a4d8","resolution":{"observed_at":"2026-05-23T00:32:18.250901Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-08-07T05:35:50.906801Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding.arXiv preprint arXiv:2109.14084, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.07600","last_updated":"2025-06-09T10:00:54Z","snapshot_observed_at":"2026-08-07T05:27:38.812324Z","submitted_at":"2025-06-09T10:00:54Z","title":"SceneRAG: Scene-level Retrieval-Augmented Generation for Video Understanding","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T05:35:50.906801Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2506.07600"},"observation_digest":"sha256:e5d39f5d8379fb06e686b867c288cc9ee5794b15099308ab3fe24d5fad703967","observation_id":"bbecc0ac-db34-4231-bf89-b3a95d52059a","resolution":{"observed_at":"2026-08-07T05:35:50.906801Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-08-07T04:09:00.365272Z","title":"Videoclip: Contrastive pre-training for zero- shot video-text understanding.arXiv preprint arXiv:2109.14084, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.14827","last_updated":"2025-06-13T13:39:53Z","snapshot_observed_at":"2026-08-07T07:12:40.611388Z","submitted_at":"2025-06-13T13:39:53Z","title":"DAVID-XR1: Detecting AI-Generated Videos with Explainable Reasoning","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-07T04:09:00.365272Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2506.14827"},"observation_digest":"sha256:530a8829ab458b4d27896a29a020932454ceef79884a2218a6fe161a1d44c82f","observation_id":"a2caf74b-d8a6-43da-894a-6d4a97a04b8b","resolution":{"observed_at":"2026-08-07T04:09:00.365272Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-08-06T23:47:35.139511Z","title":"Videoclip: Contrastive pre -training for zero -shot video-text understanding,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.16009","last_updated":"2025-06-19T04:03:58Z","snapshot_observed_at":"2026-08-06T23:42:15.552437Z","submitted_at":"2025-06-19T04:03:58Z","title":"Bridging Brain with Foundation Models through Self-Supervised Learning","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-06T23:47:35.139511Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2506.16009"},"observation_digest":"sha256:e4fe79af52144f664aa7bbb8f354f1a0da1d910952986efc4007a385fe10f44c","observation_id":"41ba384c-a049-4610-a21c-093d0006747c","resolution":{"observed_at":"2026-08-06T23:47:35.139511Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":"2109.14084","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-07-04T06:39:37.380967Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":"509d3b12-48dd-4984-9d28-a28d917f0b1c","year":2021},"citing_paper":{"arxiv_id":"2507.04590","last_updated":"2025-07-07T00:51:57Z","snapshot_observed_at":"2026-08-02T08:05:44.477432Z","submitted_at":"2025-07-07T00:51:57Z","title":"VLM2Vec-V2: Advancing Multimodal Embedding for Videos, Images, and Visual Documents","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-18T14:10:14.929207Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2507.04590"},"observation_digest":"sha256:7413aed573c239e19b34607b5c3683105451ead599634e2e2577e80b026978e9","observation_id":"61529d6c-d42b-45a7-b6ba-b13560387e97","resolution":{"observed_at":"2026-05-18T14:10:15.136859Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-08-06T17:08:56.989600Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.11892","last_updated":"2025-07-16T04:15:06Z","snapshot_observed_at":"2026-08-06T16:56:43.650358Z","submitted_at":"2025-07-16T04:15:06Z","title":"From Coarse to Nuanced: Cross-Modal Alignment of Fine-Grained Linguistic Cues and Visual Salient Regions for Dynamic Emotion Recognition","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-06T17:08:56.989600Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2507.11892"},"observation_digest":"sha256:f23e7f9c542b9f655382cace220694c76a51e3911f2a7977237dea9cc743cf7e","observation_id":"0ff0bbb6-eb8a-41d8-ba58-e1eb879b040d","resolution":{"observed_at":"2026-08-06T17:08:56.989600Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-08-06T15:07:31.643197Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.16716","last_updated":"2025-07-22T15:54:53Z","snapshot_observed_at":"2026-08-07T01:24:30.524241Z","submitted_at":"2025-07-22T15:54:53Z","title":"Enhancing Remote Sensing Vision-Language Models Through MLLM and LLM-Based High-Quality Image-Text Dataset Generation","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T15:07:31.643197Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2507.16716"},"observation_digest":"sha256:f71926b7448be931e1a4c34a8117b5f8634385bafc4ae3dbedf05e3ae2562b56","observation_id":"4b2d807e-fe8f-405d-9432-e9dd4bc960fb","resolution":{"observed_at":"2026-08-06T15:07:31.643197Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-08-06T13:24:30.193705Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.20740","last_updated":"2025-07-28T11:46:35Z","snapshot_observed_at":"2026-08-06T13:24:19.529284Z","submitted_at":"2025-07-28T11:46:35Z","title":"Implicit Counterfactual Learning for Audio-Visual Segmentation","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-06T13:24:30.193705Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2507.20740"},"observation_digest":"sha256:70da2cf0f87433fce341599466e0438e4ef29f7bb3a2013d8f701ecfbfc314b1","observation_id":"2f93e96f-c19c-41b4-8489-c070a09d9865","resolution":{"observed_at":"2026-08-06T13:24:30.193705Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-08-06T12:55:40.226378Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.21353","last_updated":"2025-07-28T21:46:05Z","snapshot_observed_at":"2026-08-07T07:17:06.405378Z","submitted_at":"2025-07-28T21:46:05Z","title":"Group Relative Augmentation for Data Efficient Action Detection","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T12:55:40.226378Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2507.21353"},"observation_digest":"sha256:a3cc5e2a252db0d4eefae2f7c043b2f1b3b69d8586cc820d5958b29437963497","observation_id":"90e63cc7-bf95-4ea5-aa75-fa1289e5b836","resolution":{"observed_at":"2026-08-06T12:55:40.226378Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-08-06T11:46:45.517124Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.22431","last_updated":"2025-07-30T07:21:36Z","snapshot_observed_at":"2026-08-06T11:46:43.177001Z","submitted_at":"2025-07-30T07:21:36Z","title":"HQ-CLIP: Leveraging Large Vision-Language Models to Create High-Quality Image-Text Datasets and CLIP Models","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T11:46:45.517124Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2507.22431"},"observation_digest":"sha256:d7fe6719f23a633269013a68f1bb3b8789b45282187c2676b5391fcc5b389cb2","observation_id":"2335abde-285a-4cac-8447-b1a2c9cbfa25","resolution":{"observed_at":"2026-08-06T11:46:45.517124Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":"2109.14084","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-07-04T06:39:37.380967Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":"509d3b12-48dd-4984-9d28-a28d917f0b1c","year":2021},"citing_paper":{"arxiv_id":"2508.06964","last_updated":"2026-04-12T08:38:14Z","snapshot_observed_at":"2026-08-06T08:43:14.125473Z","submitted_at":"2025-08-09T12:20:13Z","title":"Adversarial Video Promotion Against Text-to-Video Retrieval","version":3},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-19T00:05:07.182361Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2508.06964"},"observation_digest":"sha256:c498b34eb59e23b37199090ccaf85b9333fd5f5607309b472087ac2c1b5d612e","observation_id":"330c954d-ded8-446f-a47b-22ddfd931cff","resolution":{"observed_at":"2026-05-19T00:06:55.166881Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-08-05T18:50:58.248635Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2508.14039","last_updated":"2025-08-19T17:59:39Z","snapshot_observed_at":"2026-08-05T18:50:55.444502Z","submitted_at":"2025-08-19T17:59:39Z","title":"Beyond Simple Edits: Composed Video Retrieval with Dense Modifications","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-05T18:50:58.248635Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2508.14039"},"observation_digest":"sha256:ac0d9b984201467ba110adef85357ae6a6821c2897943c102d620e24f53834b0","observation_id":"d6b50866-ab14-445b-a0f7-1b0e75e07e7a","resolution":{"observed_at":"2026-08-05T18:50:58.248635Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-08-04T19:37:42.187507Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.09151","last_updated":"2026-06-08T07:16:30Z","snapshot_observed_at":"2026-08-06T12:45:51.887065Z","submitted_at":"2025-09-11T05:06:30Z","title":"Video Understanding by Design: How Datasets Shape Video Models","version":2},"reference_index":224,"source":"pdf_text","source_observed_at":"2026-08-04T19:37:42.187507Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2509.09151"},"observation_digest":"sha256:d94ec914457e09268714fb4bad0b238f9279ce3bbe9204e0021855acc025ab1d","observation_id":"85192c1e-362f-4e51-b9a8-049c42e00291","resolution":{"observed_at":"2026-08-04T19:37:42.187507Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":"2109.14084","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-07-04T06:39:37.380967Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":"509d3b12-48dd-4984-9d28-a28d917f0b1c","year":2021},"citing_paper":{"arxiv_id":"2511.12034","last_updated":"2026-05-12T07:28:48Z","snapshot_observed_at":"2026-08-02T17:34:14.031307Z","submitted_at":"2025-11-15T05:01:43Z","title":"Calibrated Multimodal Representation Learning with Missing Modalities","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-17T22:08:07.217659Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2511.12034"},"observation_digest":"sha256:c2ae438135cea35dc04f3d990a178a13a5738bdd7a1976ed289d9af9395e2df2","observation_id":"9761cf3d-85dc-4d2b-94c8-8d97a3f83eff","resolution":{"observed_at":"2026-05-17T22:10:22.692713Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":"2109.14084","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-07-04T06:39:37.380967Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":"509d3b12-48dd-4984-9d28-a28d917f0b1c","year":2021},"citing_paper":{"arxiv_id":"2512.03963","last_updated":"2026-04-14T11:28:58Z","snapshot_observed_at":"2026-07-31T23:59:38.983651Z","submitted_at":"2025-12-03T16:57:00Z","title":"TempR1: Improving Temporal Understanding of MLLMs via Temporal-Aware Multi-Task Reinforcement Learning","version":3},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-05-17T02:18:21.718091Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2512.03963"},"observation_digest":"sha256:48176102e64e2846a6bb69bad7bc48c7856eff34b82c638cc5e5d3d7ff1e4893","observation_id":"a30d6193-0b77-4a33-8a28-fdcf02fa13f9","resolution":{"observed_at":"2026-05-17T02:18:52.335434Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":"2109.14084","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-07-04T06:39:37.380967Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":"509d3b12-48dd-4984-9d28-a28d917f0b1c","year":2021},"citing_paper":{"arxiv_id":"2512.13511","last_updated":"2026-05-01T11:23:38Z","snapshot_observed_at":"2026-07-06T22:39:04.164003Z","submitted_at":"2025-12-15T16:38:59Z","title":"Adapting MLLMs for Nuanced Video Retrieval","version":3},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-05-16T22:20:09.051957Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2512.13511"},"observation_digest":"sha256:e4727ffad57c3d0c6274bed1cda6a403d997fc35eea20a67c5c1e852cc46046b","observation_id":"51e029eb-ce1e-4431-9dbe-a30d6a95526c","resolution":{"observed_at":"2026-05-16T22:21:18.921271Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":"2109.14084","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-07-04T06:39:37.380967Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":"509d3b12-48dd-4984-9d28-a28d917f0b1c","year":2021},"citing_paper":{"arxiv_id":"2602.02513","last_updated":"2026-05-19T07:10:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-01-23T06:39:01Z","title":"Learning ORDER-Aware Multimodal Representations for Composite Materials Design","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-21T14:36:44.656655Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2602.02513"},"observation_digest":"sha256:318bb0e54da44b1a011039a31a87595d02c09adc506d68f476377e6e4d063fdd","observation_id":"9d82ce30-3dc5-4cb9-8455-d148e1611781","resolution":{"observed_at":"2026-05-21T14:40:14.550097Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":"2109.14084","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-07-04T06:39:37.380967Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":"509d3b12-48dd-4984-9d28-a28d917f0b1c","year":2021},"citing_paper":{"arxiv_id":"2603.09921","last_updated":"2026-07-02T16:25:34Z","snapshot_observed_at":"2026-07-14T23:55:23.688065Z","submitted_at":"2026-03-10T17:18:53Z","title":"WikiCLIP: An Efficient Contrastive Baseline for Open-domain Visual Entity Recognition","version":3},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-15T13:11:54.384284Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2603.09921"},"observation_digest":"sha256:62a0d32b5e4b60dd3ab73f40df634af0fe846cd546606307df27368d08e42f05","observation_id":"98a3f7aa-d2a1-49c0-a7f6-9169612aa447","resolution":{"observed_at":"2026-05-15T13:15:50.570008Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-07-14T23:55:24.006436Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding.arXiv preprint arXiv:2109.14084, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2603.09921","last_updated":"2026-07-02T16:25:34Z","snapshot_observed_at":"2026-07-14T23:55:23.688065Z","submitted_at":"2026-03-10T17:18:53Z","title":"WikiCLIP: An Efficient Contrastive Baseline for Open-domain Visual Entity Recognition","version":4},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-07-14T23:55:24.006436Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2603.09921"},"observation_digest":"sha256:04b856800fdf9bb4be0613dc2ee273b85da01662795959a3190d6013c2230da4","observation_id":"e5989aae-07c8-4169-a9bf-75f9fcec2bff","resolution":{"observed_at":"2026-07-14T23:55:24.006436Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-07-13T21:37:55.887477Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding.arXiv preprint arXiv:2109.14084, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2603.20190","last_updated":"2026-06-09T22:16:34Z","snapshot_observed_at":"2026-08-04T20:13:17.293244Z","submitted_at":"2026-03-20T17:59:25Z","title":"CoVR-R:Reason-Aware Composed Video Retrieval","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-07-13T21:37:55.887477Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2603.20190"},"observation_digest":"sha256:b881d942c65cf59ce114dcecf317b09e9d3538dc352f1cdd4545102ae19cfd48","observation_id":"79bc04fd-414f-4150-a5c8-710534ff86cb","resolution":{"observed_at":"2026-07-13T21:37:55.887477Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":"2109.14084","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-07-04T06:39:37.380967Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":"509d3b12-48dd-4984-9d28-a28d917f0b1c","year":2021},"citing_paper":{"arxiv_id":"2604.04875","last_updated":"2026-04-06T17:26:04Z","snapshot_observed_at":"2026-07-06T22:53:46.999911Z","submitted_at":"2026-04-06T17:26:04Z","title":"DIRECT: Video Mashup Creation via Hierarchical Multi-Agent Planning and Intent-Guided Editing","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-10T19:44:52.768021Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2604.04875"},"observation_digest":"sha256:aace3e29c972be7ddd44f032cbc5ce36658315748f50ac9755e618c1fac3e132","observation_id":"c3f96828-e72e-4f38-953b-6c85b0fefb02","resolution":{"observed_at":"2026-05-10T22:30:53.943204Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":"2109.14084","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-07-04T06:39:37.380967Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":"509d3b12-48dd-4984-9d28-a28d917f0b1c","year":2021},"citing_paper":{"arxiv_id":"2604.08337","last_updated":"2026-04-09T15:10:25Z","snapshot_observed_at":"2026-08-03T10:29:21.842229Z","submitted_at":"2026-04-09T15:10:25Z","title":"InstAP: Instance-Aware Vision-Language Pre-Train for Spatial-Temporal Understanding","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-05-10T18:06:33.139310Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2604.08337"},"observation_digest":"sha256:a562528f1e57fe0f3928a6bd11e4a5ab0b05f4266bf65d2f5f164f164e639b62","observation_id":"a286e855-f8a7-4198-b58a-1c9456c2bb14","resolution":{"observed_at":"2026-05-11T05:30:58.326399Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":"2109.14084","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-07-04T06:39:37.380967Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":"509d3b12-48dd-4984-9d28-a28d917f0b1c","year":2021},"citing_paper":{"arxiv_id":"2604.14684","last_updated":"2026-05-26T13:10:47Z","snapshot_observed_at":"2026-07-12T20:05:14.844082Z","submitted_at":"2026-04-16T06:40:44Z","title":"DETR-ViP: Detection Transformer with Robust Discriminative Visual Prompts","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-10T12:07:21.203513Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2604.14684"},"observation_digest":"sha256:e869acbdb8223788f021b1b6385f33444ce251a6900e61dce7c3726faa6b24a0","observation_id":"bf843ac4-63bf-4441-91ca-748db06036a3","resolution":{"observed_at":"2026-05-10T12:10:21.931008Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-07-12T20:05:20.702092Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding.arXiv preprint arXiv:2109.14084,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2604.14684","last_updated":"2026-05-26T13:10:47Z","snapshot_observed_at":"2026-07-12T20:05:14.844082Z","submitted_at":"2026-04-16T06:40:44Z","title":"DETR-ViP: Detection Transformer with Robust Discriminative Visual Prompts","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-07-12T20:05:20.702092Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2604.14684"},"observation_digest":"sha256:b465b76f03d2a07eb952b530fa2b8f287cc7db25e8eb8728ad3daf56650f3bcb","observation_id":"ee162317-38b2-4cf0-8ca7-b03ee279b6d7","resolution":{"observed_at":"2026-07-12T20:05:20.702092Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":"2109.14084","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-07-04T06:39:37.380967Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":"509d3b12-48dd-4984-9d28-a28d917f0b1c","year":2021},"citing_paper":{"arxiv_id":"2605.00051","last_updated":"2026-04-29T15:29:19Z","snapshot_observed_at":"2026-08-02T05:45:46.015951Z","submitted_at":"2026-04-29T15:29:19Z","title":"Learning from the Unseen: Generative Data Augmentation for Geometric-Semantic Accident Anticipation","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-09T20:04:39.623342Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2605.00051"},"observation_digest":"sha256:2c467695ccd2a70712741c36e8fc9dd91517686975c8506388c6ceaca81a60b2","observation_id":"e05b8581-ffea-4cf1-8fd9-ae8478f83ced","resolution":{"observed_at":"2026-05-11T15:26:08.280521Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":"2109.14084","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-07-04T06:39:37.380967Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":"509d3b12-48dd-4984-9d28-a28d917f0b1c","year":2021},"citing_paper":{"arxiv_id":"2605.21059","last_updated":"2026-05-20T11:44:01Z","snapshot_observed_at":"2026-07-06T23:31:32.577242Z","submitted_at":"2026-05-20T11:44:01Z","title":"Multimodal LLMs under Pairwise Modalities","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-05-21T05:37:56.792564Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2605.21059"},"observation_digest":"sha256:c0cca381ab9c25ce1ae9af52ccda33da24b1411077556e31ea867aab20008139","observation_id":"6d4afa95-e025-4962-a74a-fc809fa40d2b","resolution":{"observed_at":"2026-05-21T05:39:40.686881Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":"2109.14084","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-07-04T06:39:37.380967Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":"509d3b12-48dd-4984-9d28-a28d917f0b1c","year":2021},"citing_paper":{"arxiv_id":"2606.02273","last_updated":"2026-06-01T13:59:46Z","snapshot_observed_at":"2026-07-06T23:42:44.661608Z","submitted_at":"2026-06-01T13:59:46Z","title":"Vision-language Models for Driver Monitoring Systems: A Driver Activity Description Dataset","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-06-28T15:29:51.424572Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2606.02273"},"observation_digest":"sha256:008547a0b769ba77e39d6b2b1220dcf886c03caac105a1ce38153d6ebf55dd4c","observation_id":"b15f4643-7da8-4008-bbbf-485042c8b020","resolution":{"observed_at":"2026-07-01T22:16:16.925691Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":"2109.14084","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-07-04T06:39:37.380967Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":"509d3b12-48dd-4984-9d28-a28d917f0b1c","year":2021},"citing_paper":{"arxiv_id":"2606.21734","last_updated":"2026-06-19T20:43:49Z","snapshot_observed_at":"2026-08-05T18:05:51.515234Z","submitted_at":"2026-06-19T20:43:49Z","title":"HPP: Hierarchical Programmatic Probing for Long Video Understanding by Decoupling Perception and Reasoning","version":1},"reference_index":97,"source":"arxiv_source","source_observed_at":"2026-06-26T14:19:53.450263Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2606.21734"},"observation_digest":"sha256:43edbb428f77bfa3cc98e861eb9d3d2d8392c50b78b780b3ef7a6bb0a8378955","observation_id":"8acc79c7-1e61-4a3c-85fb-d94956ca9fc6","resolution":{"observed_at":"2026-07-04T06:39:37.382403Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2109.14084/citation-record","integrity":"/paper/2109.14084/integrity","json":"/paper/2109.14084/citation-record.json","paper":"/paper/2109.14084"},"outbound":[],"paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 35 inbound Pith citation observations for arXiv:2109.14084."}