{"as_of":"2026-08-13T08:15:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:3da2ae95e0e48b5816562b09b06ea392ceeba780d7e1409d716400eb9bbc3877","coverage":[{"denominator":95,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":95,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-10T23:12:06.172447Z","state":"measured"},{"denominator":95,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":95,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-13T06:32:02.005865+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2412.20964/citation-record","integrity":"/paper/2412.20964/integrity","json":"/paper/2412.20964/citation-record.json","paper":"/paper/2412.20964"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:05.723321Z","title":"Parallel Vertex Diffusion for Unified Visual Grounding,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.723321Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:bc59639d47763c6ef2e3be5cf25a39042f71b0269cae868d7adbc29fb3b1d98b","observation_id":"580181f6-c219-467d-8a29-3675c4e7e910","resolution":{"observed_at":"2026-08-10T23:12:05.723321Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:05.729502Z","title":"Align and Prompt: Video-and-Language Pre-training with Entity Prompts,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.729502Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:8e893d6b81cc557c94c8c742fc04817dbb5bd3d38c3b1b101b04b1ca844d1aef","observation_id":"e2caa0d2-1939-4c13-b5b1-490a56ac8513","resolution":{"observed_at":"2026-08-10T23:12:05.729502Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:05.734526Z","title":"Expectation-Maximization Contrastive Learning for Compact Video-and-Language Representations,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.734526Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:d242522986d4e740083822fb06c69df9910923b1d4718ef19ced496e6f622d38","observation_id":"e24e2e46-65b9-48d4-b2b7-2d2faf469f31","resolution":{"observed_at":"2026-08-10T23:12:05.734526Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:05.740405Z","title":"FreestyleRet: Retrieving Images from Style-Diversified Queries,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.740405Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:bb9497086ad59f792658d5155858b5a526f5c8305b8f4c5b862607a9e33d6424","observation_id":"01d8a163-76f5-466d-913b-9e3a768d01db","resolution":{"observed_at":"2026-08-10T23:12:05.740405Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:05.746306Z","title":"Many Hands Make Light Work: Transferring Knowledge from Auxiliary Tasks for Video- Text Retrieval,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.746306Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:d50135d9fe057f0cf1f1f9699cf79404e78a4de1336929a7626cef3ad37751bc","observation_id":"b1aaa9f2-4362-4779-9054-8aa9998607cd","resolution":{"observed_at":"2026-08-10T23:12:05.746306Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:05.751937Z","title":"Dual Encoding for Video Retrieval by Text,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.751937Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:49ad52952e4e6b2c17264269743beafac5dd8a95d14f980d39da32ec3acb8579","observation_id":"29ef21c2-3c08-41e0-80e4-ea9205a68081","resolution":{"observed_at":"2026-08-10T23:12:05.751937Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:05.757710Z","title":"Temporal Alignment Networks for Long-term Video,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.757710Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:d65f7705bcfd914e27315d69e2883e231697731b2ff0441e0e71cbb56678d5f2","observation_id":"b8abaa5a-d856-47b7-ad00-344a54ca18ef","resolution":{"observed_at":"2026-08-10T23:12:05.757710Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:05.762521Z","title":"Dif- fusionRet: Generative Text-Video Retrieval with Diffusion Model,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.762521Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:7ac462ba91ebead072a4cc842b4ecb90798402a59ddbc2df07acf6b346fe00b5","observation_id":"9a402677-488d-4e60-b5c4-0642ff5be8a1","resolution":{"observed_at":"2026-08-10T23:12:05.762521Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:05.767544Z","title":"An axiomatic approach to the concept of interaction among players in cooperative games,","venue":null,"work_id":null,"year":1999},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.767544Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:a9494131f66478fe0063439c69ec501e4a7435d82cdd527be6fb798ae7dc4a83","observation_id":"fadebf4e-91ec-441e-8b23-2b5dbf0299c1","resolution":{"observed_at":"2026-08-10T23:12:05.767544Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:05.772384Z","title":"Weighted Banzhaf power and interaction indexes through weighted approximations of games,","venue":null,"work_id":null,"year":2011},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.772384Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:9dbee6c781fcb34e83a04a5520624b08909c80ccbbc96e8ce09f698dd144b4cc","observation_id":"dd351e5b-f8c8-4aeb-9b84-439383506c24","resolution":{"observed_at":"2026-08-10T23:12:05.772384Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:05.776983Z","title":"Video-Text as Game Players: Hierarchical Banzhaf Interaction for Cross-Modal Representation Learning,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.776983Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:1e20fb85e0592d669618d695e21404158c7667de88928a4843be8fd71152b92e","observation_id":"f4cecc72-b5a1-4a92-9794-f533cd60633d","resolution":{"observed_at":"2026-08-10T23:12:05.776983Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:05.782331Z","title":"MSR-VTT: A Large Video Description Dataset for Bridging Video and Language,","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.782331Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:d686b27a9e3c63f87e27e0b5d9922bf52c2f5a807ff2d784c79cdbf944aebd88","observation_id":"dc667b79-58bb-48a1-b6d1-09c9a51b3d9e","resolution":{"observed_at":"2026-08-10T23:12:05.782331Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:05.787122Z","title":"Dense- Captioning Events in Videos,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.787122Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:9ae6bf701e073cd3998c1134b716c874b3801b1c54cf0af7d5d5ce956192b97e","observation_id":"8257d3ac-c9d2-4e89-9ffd-6045984ec663","resolution":{"observed_at":"2026-08-10T23:12:05.787122Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:05.791722Z","title":"Localizing Moments in Video with Natural Lan- guage,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.791722Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:025647a6e47f1affb7b025561cf61fe3c58bb977b23efdf9ad1944e51294d789","observation_id":"fce22814-90f7-495f-9113-89221bb66181","resolution":{"observed_at":"2026-08-10T23:12:05.791722Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:05.796656Z","title":"Video Question Answering via Gradually Refined Attention over Appearance and Motion,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.796656Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:74c86c67548d344aec503aa76738c77c0436e569049962882b4baf40c3a8e4be","observation_id":"de3403c7-4884-4a80-9f49-c8750cb0a0bc","resolution":{"observed_at":"2026-08-10T23:12:05.796656Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:07.453629Z","title":"ActivityNet-QA: A Dataset for Understanding Complex Web Videos via Question Answering,","venue":null,"work_id":"dcc26974-e648-4768-850b-471dd8b821f7","year":2019},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.801152Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:bc15ed38d5788f80c7966671d7d07e9288637d813dadebff22f5fd9c5102144d","observation_id":"b274597a-7c10-46af-8fe2-9eb2b24caf6b","resolution":{"observed_at":"2026-08-10T23:12:07.459256Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:07.436759Z","title":"Universal Weight- ing Metric Learning for Cross-Modal Retrieval,","venue":null,"work_id":"5339f30b-43b4-4765-a7ee-a8910e3b76f5","year":2021},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.805550Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:e8f2e3be92b31e4833154d52b93471c1bfd10de96e0444b145b875581bacc6ae","observation_id":"0f155bde-624d-470e-ab62-1c829b685aab","resolution":{"observed_at":"2026-08-10T23:12:07.442279Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:07.420427Z","title":"Weakly- Supervised 3D Spatial Reasoning for Text-Based Visual Question Answering,","venue":null,"work_id":"4eae35e8-af51-410d-8ce7-8781cf4cdb51","year":2023},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.810678Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:d3ff9ec2b1a43ba4749581be1a50e2658db6517f8930625cad7515854173b16a","observation_id":"826b2159-8ae8-475c-9961-5f410fb49b6c","resolution":{"observed_at":"2026-08-10T23:12:07.426202Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:07.403309Z","title":"Revisiting the ‘Video’ in Video-Language Understand- ing,","venue":null,"work_id":"c440bdc4-2f54-4b9d-957d-9b3d3f823068","year":2022},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.815304Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:98af51736e3ac8f69280c0909e23c7d69e96c12e5c948f911d3088961d09615a","observation_id":"187ba5fb-4864-40ca-860e-08adcb7f871e","resolution":{"observed_at":"2026-08-10T23:12:07.409207Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:07.386353Z","title":"Fine-Grained Semantically Aligned Vision-Language Pre-Training,","venue":null,"work_id":"f578d473-a1c6-4268-817f-9460a81eaad6","year":2022},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.820300Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:8c52fca250ed04bb44dbb58932ef9492130ccfbd39dfaecf0a22770c3813e76e","observation_id":"5131f6e0-6939-480c-90d9-d4f522313cd6","resolution":{"observed_at":"2026-08-10T23:12:07.391727Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:07.370860Z","title":"Chat- UniVi: Unified Visual Representation Empowers Large Language Models with Image and Video Understanding,","venue":null,"work_id":"c9614fa5-d142-43c3-977c-1970b697de82","year":2024},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.824893Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:00da54a6c07fdcef2510355eca43190ea52d835aeffc57120846881ed5f5be87","observation_id":"ed08a169-cf2c-4d57-a05a-1acbe1425b0c","resolution":{"observed_at":"2026-08-10T23:12:07.376631Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:07.354002Z","title":"LanguageBind: Extending Video-Language Pretraining to N- modality by Language-based Semantic Alignment,","venue":null,"work_id":"c23f97df-214b-4d85-a62b-6bf17e1b4916","year":2024},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.829969Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:c4a3568d422ba23eb565a4fdd18f9a121a122ab7136860befc053ccc21094639","observation_id":"be87c5ae-cc08-40cc-b5e0-d85657143a3c","resolution":{"observed_at":"2026-08-10T23:12:07.360014Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:07.337491Z","title":"Decoupled peak property learning for efficient and interpretable ecd spectra prediction,","venue":null,"work_id":"fa78b7d2-73ab-44fc-8416-9d216fb0d18d","year":2024},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.835414Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:9feab421cb029f774f79219d1b600b159103cb6d4c5831b209691b052a49b3d0","observation_id":"ba5899eb-b8d5-44db-9f26-2564ea2e10f4","resolution":{"observed_at":"2026-08-10T23:12:07.342699Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.10440","last_updated":"2025-07-21T03:53:30Z","snapshot_observed_at":"2026-08-07T09:36:29.319006Z","submitted_at":"2024-11-15T18:58:31Z","title":"LLaVA-CoT: Let Vision Language Models Reason Step-by-Step","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.10440","snapshot_observed_at":"2026-08-10T23:12:05.839893Z","title":"LLaVA-o1: Let Vision Language Models Reason Step-by-Step,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.839893Z"},"links":{"cited_paper":"/paper/2411.10440","citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:2217a633658841fcf5278d37d03ac015b0be344b12ee090c073c6693afe2f1dd","observation_id":"85c7ebb0-03aa-4346-af36-21c2b3c6941c","resolution":{"observed_at":"2026-08-10T23:12:05.839893Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20224","last_updated":"2024-12-06T11:34:57Z","snapshot_observed_at":"2026-08-12T23:55:32.210154Z","submitted_at":"2024-05-29T04:59:27Z","title":"EvaGaussians: Event Stream Assisted Gaussian Splatting from Blurry Images","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20224","snapshot_observed_at":"2026-08-10T23:12:05.845207Z","title":"Evagaussians: Event stream assisted gaussian splatting from blurry images,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.845207Z"},"links":{"cited_paper":"/paper/2405.20224","citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:6dadb987295fe59db67991e09e2f6954687b9e2e911fce14cbbf31e0f9649d90","observation_id":"0084ff6b-a2bd-4067-9791-3afa3253ddd3","resolution":{"observed_at":"2026-08-10T23:12:05.845207Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.19548","last_updated":"2024-07-28T17:58:35Z","snapshot_observed_at":"2026-08-12T23:13:20.814777Z","submitted_at":"2024-07-28T17:58:35Z","title":"Cycle3D: High-quality and Consistent Image-to-3D Generation via Generation-Reconstruction Cycle","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.19548","snapshot_observed_at":"2026-08-10T23:12:05.850062Z","title":"Cycle3d: High-quality and consistent image-to- 3d generation via generation-reconstruction cycle,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.850062Z"},"links":{"cited_paper":"/paper/2407.19548","citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:27ead6831b3c168bd046c61bbcedb5a2580eb434629c5285e0083449d4d5d278","observation_id":"7ac6680f-dac9-4815-81fc-e833b5061aed","resolution":{"observed_at":"2026-08-10T23:12:05.850062Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:07.321417Z","title":"Repaint123: Fast and high-quality one image to 3d generation with progressive controllable repainting,","venue":null,"work_id":"99be8ad8-5ffe-460d-a683-f8a04007d240","year":2025},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.854989Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:8ab2713f55e1617f6c720d2d9ce07bf0e19251303ad309108cbb52710f2da0be","observation_id":"233a77f0-5372-4513-8faf-0789cd268a65","resolution":{"observed_at":"2026-08-10T23:12:07.326839Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15321","last_updated":"2025-03-19T06:16:54Z","snapshot_observed_at":"2026-08-11T15:20:00.371059Z","submitted_at":"2024-12-19T18:59:36Z","title":"Next Patch Prediction for Autoregressive Visual Generation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15321","snapshot_observed_at":"2026-08-10T23:12:05.860105Z","title":"Next patch prediction for autoregressive visual generation,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.860105Z"},"links":{"cited_paper":"/paper/2412.15321","citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:7e82ae17577dcf5f64301c5c7884edad42710767a7d2259ad2723aa7ffe5203a","observation_id":"1457285d-4191-414b-bd55-2cec95ab0033","resolution":{"observed_at":"2026-08-10T23:12:05.860105Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:07.305628Z","title":"Learning the Best Pooling Strategy for Visual Semantic Embedding,","venue":null,"work_id":"d1d98418-c867-46d8-acda-89e2a68b7af7","year":2021},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.864794Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:b8ffc2be937a54188089ae26a35e81f3faac267e8d659b9b9c48367fdba39764","observation_id":"06d71f36-46ef-48fe-8df3-c76f22863b9a","resolution":{"observed_at":"2026-08-10T23:12:07.310826Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:07.289303Z","title":"DGL: Dynamic Global- Local Prompt Tuning for Text-Video Retrieval,","venue":null,"work_id":"a85a1fb6-b3a6-4a17-8664-3783dce01cf3","year":2024},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.869660Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:48a7e96b1007c1bdf3b1231a4f334c961daa65c5f63d8782edc7114eaf076dcb","observation_id":"5c620472-2588-440c-b775-3ea6db513a70","resolution":{"observed_at":"2026-08-10T23:12:07.294215Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:07.274032Z","title":"Learning Transferable Visual Models From Natural Language Supervision,","venue":null,"work_id":"016d2b33-0198-4169-975c-1758570e9982","year":2021},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.873998Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:db52f767ba899d097fa2c95026f26911f8af49910e3ebbd74a04db67a69dc453","observation_id":"60c8fc4d-61b7-47e8-8edf-ed2d186a6c97","resolution":{"observed_at":"2026-08-10T23:12:07.278911Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:07.257632Z","title":"ViSTA: Vision and Scene Text Aggregation for Cross-Modal Retrieval,","venue":null,"work_id":"23e0bbb0-bcb4-416e-bd76-751e022ee0f6","year":2022},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.878311Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:a51421a7f00d3664a204c82e632f615defd2758e0600b01f559575f0c89ae960","observation_id":"46fe8fa6-37fc-4019-8dcd-858642da83cd","resolution":{"observed_at":"2026-08-10T23:12:07.263624Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:07.242865Z","title":"SUTD-TrafficQA: A Question An- swering Benchmark and an Efficient Network for Video Reasoning over Traffic Events,","venue":null,"work_id":"8d881734-8348-433c-b1a9-6bc3b180d072","year":2021},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.883444Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:57cebb7a360b3953c563b1f3e564ace769aee7e548f8fdbab6d7478addb58b20","observation_id":"bc757e8e-5304-499b-b4f7-8b5485bbea67","resolution":{"observed_at":"2026-08-10T23:12:07.247617Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:07.227774Z","title":"Hierarchical Con- ditional Relation Networks for Video Question Answering,","venue":null,"work_id":"e2ead709-3c3d-4ce8-a8b0-d4c78ef74483","year":2020},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.887935Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:cf1cd5afb255bcdec2dcc2dfa8e7df20a60dfc9e3740dfb0d590f569d4a889e7","observation_id":"aba342a8-9f5d-4cfe-88f6-5fa7c2d1eac4","resolution":{"observed_at":"2026-08-10T23:12:07.232752Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:07.212628Z","title":"Video Question Answering: Datasets, Algorithms and Challenges,","venue":null,"work_id":"678cec68-92d7-4780-ae65-8d3555046e4f","year":2022},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.892548Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:60ec832f61d092c4b5aa008d41103981c81cd755d104673f72d9cf9afee44510","observation_id":"d05302cd-1063-4a5e-9a41-d0d265532e07","resolution":{"observed_at":"2026-08-10T23:12:07.217463Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:07.194804Z","title":"Less Is More: ClipBERT for Video-and-Language Learning via Sparse Sampling,","venue":null,"work_id":"b344dfa9-2399-425f-b1e4-637d54fac8a4","year":2021},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.897229Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:93ba7575e9e9fd0d7e9f5e413b168eb935695ffd87aae13e84197471f7d46ad8","observation_id":"cecd60ef-9e7a-4f28-9cc1-9663220d202e","resolution":{"observed_at":"2026-08-10T23:12:07.200092Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:07.178590Z","title":"Video Question Answering with Iterative Video-Text Co- Tokenization,","venue":null,"work_id":"01a5f1f5-6736-4f41-a413-45b2daff030c","year":2022},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.901654Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:8d0e2ea6f0bfb12ca69483fc84484bbba03b8064fc6c74d7a867e53fb4f2094e","observation_id":"3aee8a22-e517-41c6-8159-c847b1d072cd","resolution":{"observed_at":"2026-08-10T23:12:07.183820Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:07.161896Z","title":"Frozen in Time: A Joint Video and Image Encoder for End-to-End Retrieval,","venue":null,"work_id":"35596dfa-8ab2-414b-85f7-372aab30031f","year":2021},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.906018Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:c6924b41789d96e278100982e7f3786d1ca6489546632c6ee77600808e08f53e","observation_id":"3d53e862-8086-4691-8574-2ff90815e13e","resolution":{"observed_at":"2026-08-10T23:12:07.167383Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:07.146299Z","title":"Multilingual Multimodal Pre-training for Zero-Shot Cross- Lingual Transfer of Vision-Language Models,","venue":null,"work_id":"9ff4de36-962f-4935-91f2-ef0ccb05ecf3","year":2021},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.910639Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:f0693dcfac1f1f85bff490444330beb93355714b0d33a684035b158034e27f25","observation_id":"6f94017d-14a8-43b1-9f46-e0e0993c33f0","resolution":{"observed_at":"2026-08-10T23:12:07.151121Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:07.131478Z","title":"Jointly Localizing and Describing Events for Dense Video Captioning,","venue":null,"work_id":"dca99c5f-6536-40ba-9734-6eb3af368272","year":2018},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.915159Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:fa26e9bf861eb79387b9ac449ffaa04b89acbd40c42205496014df10747940df","observation_id":"06dea930-9905-4701-b330-7647b8ec12b1","resolution":{"observed_at":"2026-08-10T23:12:07.136079Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:07.116692Z","title":"Video Captioning with Transferred Semantic Attributes,","venue":null,"work_id":"a7874fb5-4737-4a6d-82a6-007fff9b1013","year":2017},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.919598Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:d79cb27606a6c2b8099d94d1ae13c33c3037aee7123ea5b5ff776e0cc4973ab0","observation_id":"72c9e196-afda-4e0b-b89a-ba78e686ab17","resolution":{"observed_at":"2026-08-10T23:12:07.121726Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:07.100862Z","title":"Jointly Modeling Embedding and Translation to Bridge Video and Language,","venue":null,"work_id":"80d2a3a9-4edf-4c69-9499-d0660c2e0dae","year":2016},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.924199Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:7c2c1818ea6fb1f30bc675fa92e51608fa56dcca1e2ae6c218d329393213f049","observation_id":"30e5a4da-b26c-40e9-8b55-4ed14d965516","resolution":{"observed_at":"2026-08-10T23:12:07.106098Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:07.085412Z","title":"Retrieval Augmented Convolutional Encoder-Decoder Networks for Video Captioning,","venue":null,"work_id":"aff23d7f-267a-4fc7-b7cd-bfa5b8897994","year":2023},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.928690Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:9c02fc1ed5444973c19080db9645270f9ec1b76a515b97007db002e3a2030042","observation_id":"b1c5967c-c4d2-4621-b14f-94ede0299011","resolution":{"observed_at":"2026-08-10T23:12:07.090476Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:07.070012Z","title":null,"venue":null,"work_id":"4c658e3b-f7b4-46d9-ae4e-bf92b540f10e","year":2020},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.933482Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:c702ce82ed88c33f783a335f26e75745d433cc5eeecc7fa2550bdf2b47224e29","observation_id":"7ade93c9-c123-4725-b194-f045060e46e9","resolution":{"observed_at":"2026-08-10T23:12:07.074915Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:07.054723Z","title":"Optical-model po- tential in finite nuclei from Reid’s hard core interaction,","venue":null,"work_id":"717f2a95-67e6-426d-9f1b-497f7b48238c","year":1977},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.937927Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:09df04e844dfc7a734633e44f6668b42077497446cb3f9e7f26f11b2e64977da","observation_id":"07426d4a-e689-4008-9b1a-5ba4e3deb0bb","resolution":{"observed_at":"2026-08-10T23:12:07.059636Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:07.038728Z","title":"Random Shapley Forests: Cooperative Game Based Random Forests with Consistency,","venue":null,"work_id":"7ebcbafe-5630-4f95-9427-6fa87d5b184d","year":2020},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.942379Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:224f7450147e7c7c36ec5cc524a6c5a06eb27f45ed289c35a9dc958b0258bf3d","observation_id":"01e9f7a3-a58c-4079-a4e6-1bb56496094c","resolution":{"observed_at":"2026-08-10T23:12:07.044075Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:07.021482Z","title":"VL-InterpreT: An Interactive Visualization Tool for Interpreting Vision-Language Transformers,","venue":null,"work_id":"394d5543-c0dc-4cf2-a58d-a04c5b35e558","year":2022},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.947661Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:ff69ff263241fac273124750c44da452b39f26795d0d6a520280ffdf82afd2ff","observation_id":"a32ecb05-ee33-4dca-90a3-2989dc7ab9d9","resolution":{"observed_at":"2026-08-10T23:12:07.026810Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:07.004645Z","title":"Algorithmic Transparency via Quan- titative Input Influence: Theory and Experiments with Learning Systems,","venue":null,"work_id":"f0a72a0c-a4a2-4169-8a84-06176756589e","year":2016},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.952180Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:7e9b5cabb6bb519c374567b01f0bab009d2221a1292858c5c625ad8642cee3f7","observation_id":"d89eefbc-6df6-41e0-9283-65bb44626049","resolution":{"observed_at":"2026-08-10T23:12:07.010019Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:06.988677Z","title":"Text-Video Retrieval with Disentangled Conceptualiza- tion and Set-to-Set Alignment,","venue":null,"work_id":"b1f586dd-0769-4c74-8cec-09da34c6cd38","year":2023},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.956766Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:7fe5ed64e5b9060d2282437ef5c9ffb6bd77fba9aee6d1f3861613100718ed02","observation_id":"d12f0efa-572a-49ea-964e-9618a659463d","resolution":{"observed_at":"2026-08-10T23:12:06.993715Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:06.971532Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale,","venue":null,"work_id":"7166a3bc-2c97-4f17-b92c-a76af54262c3","year":2021},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.961335Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:66a92d586abb4495f46047ba65041d81a821af02802b3890fa82377e3abd8c5d","observation_id":"b4435afe-cd1d-43ca-aee0-7ed9b362b0b1","resolution":{"observed_at":"2026-08-10T23:12:06.976856Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:06.955655Z","title":"Kullback, Information Theory and Statistics","venue":null,"work_id":"b4b4b41c-7677-4c0e-9b1f-bf3072424aa1","year":1997},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.966063Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:ea7851742e302c2d505c39b4538925d601c15fafeb8d57c6a3168914505cd735","observation_id":"6b280c4e-95ac-4ae6-acd2-1ccd5c61d464","resolution":{"observed_at":"2026-08-10T23:12:06.960734Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:05.971077Z","title":"Study on density peaks clustering based on k-nearest neighbors and principal component analysis,","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.971077Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:92280b82e44d8090f2a85c0ab40aa9a01e7d28bd3cf52e06e2124fb0dfa451ae","observation_id":"e03b7460-3f52-4367-9457-1b99a1dcf37c","resolution":{"observed_at":"2026-08-10T23:12:05.971077Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:06.927747Z","title":"ACSeg: Adaptive Conceptualization for Unsuper- vised Semantic Segmentation,","venue":null,"work_id":"fe28f44f-e9d4-4a4d-af38-42d0e1ac3b68","year":2023},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.975596Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:9d1acbe91cb7199a42a87a477d3a02192b36b765b5ebf0b41a7a01fe2d130783","observation_id":"5ee34466-8277-4045-8893-f4b3911909ad","resolution":{"observed_at":"2026-08-10T23:12:06.934023Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:06.911789Z","title":"Dynam- icViT: Efficient Vision Transformers with Dynamic Token Sparsifi- cation,","venue":null,"work_id":"9beaeae2-2586-437d-9a87-7b9a94653fa9","year":2021},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.980369Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:394128bea062584d920aea6d014fd150111c6c470b2f6238a70275efbce82cdf","observation_id":"e3afa91d-1e33-4046-8e5e-f679205bacfd","resolution":{"observed_at":"2026-08-10T23:12:06.917159Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:06.896421Z","title":"Cross Modal Retrieval with Querybank Normalisation,","venue":null,"work_id":"86a63216-c01b-4dfa-9c85-9b73f1a3ee3f","year":2022},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.984936Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:178cfa5b52b4b9273b9fbc5f3ff71a350f37e815200a28a7a76d9127fbf7e849","observation_id":"ba9f6136-6bcf-4314-ba69-d67834dda4f0","resolution":{"observed_at":"2026-08-10T23:12:06.901496Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:06.880396Z","title":"Multi-modal Transformer for Video Retrieval,","venue":null,"work_id":"9e5310e4-0944-489c-ab7d-6cad4e125ad7","year":2020},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.990474Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:ef2e728f95ff21cf4881df714e005c1e8cfc33d742ed693587f18f5e0845955e","observation_id":"997268e8-2bf2-473c-b96c-eae4f7ab6974","resolution":{"observed_at":"2026-08-10T23:12:06.885504Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:06.865337Z","title":"T2VLAD: Global-Local Se- quence Alignment for Text-Video Retrieval,","venue":null,"work_id":"4beee25b-afa5-4ebf-9792-db3e4bcfb453","year":2021},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.994958Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:f6fadd6055f20a7e44e27671efc7c9317a3b50f22925f2a12226f3f03a2d45d2","observation_id":"8cce65b0-6d35-4cfb-98b6-760e8e60abd7","resolution":{"observed_at":"2026-08-10T23:12:06.870428Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:06.849489Z","title":"TEACHTEXT: CrossModal General- ized Distillation for Text-Video Retrieval,","venue":null,"work_id":"0302d7c6-3ed6-48d9-bfe9-51c3700f073d","year":2021},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:05.999483Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:6455c8bd2f88d6c8a62381c495a7a2181e907fa703ef4df7e226ace99c96da25","observation_id":"70980822-f817-4cbb-88e1-50106cc2be1d","resolution":{"observed_at":"2026-08-10T23:12:06.854606Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:06.834591Z","title":"Support-set bottlenecks for video- text representation learning,","venue":null,"work_id":"62417512-1bc1-4fe9-af7a-7ae31605a434","year":2021},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:06.004645Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:b8191d1d8f2f36d8d5eb14afb4210271d2140173e8c32e13f53ecf56870b787a","observation_id":"e55b30e8-aa31-416c-8cce-50fef8ba3127","resolution":{"observed_at":"2026-08-10T23:12:06.839480Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:06.819353Z","title":"CLIP4Clip: An Empirical Study of CLIP for End to End Video Clip Retrieval,","venue":null,"work_id":"4eb2a838-1eaa-4d4b-9530-e1344083f394","year":2022},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:06.009501Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:399420e9ee5ae018e0f1f2b35eabf9fbe109179d15fff5f5fd3ce7ce6626b3ed","observation_id":"bd42f802-5095-4e2c-80d9-bb264a531cfc","resolution":{"observed_at":"2026-08-10T23:12:06.824809Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:06.803584Z","title":"X-Pool: Cross-Modal Language-Video Attention for Text-Video Retrieval,","venue":null,"work_id":"cdcae485-e62d-45b6-849a-61eedfb4a913","year":2022},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:06.013862Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:76ad39a464dbcaccc3fd205b03724716de963f5418dd1c3ec652df173faa0b44","observation_id":"cb85fc88-9c83-4ee4-9f5d-ed059dc9c89e","resolution":{"observed_at":"2026-08-10T23:12:06.808890Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:06.788332Z","title":"TS2-Net: Token Shift and Selection Transformer for Text-Video Retrieval,","venue":null,"work_id":"21a47731-6375-4f93-b66e-6bcd376df771","year":2022},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:06.018307Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:3a4db0321d48eac4968b403f3a148ae303abac13a8e06d42beb066fdc29bb247","observation_id":"57368048-0dcc-483a-884b-ccffb54fb35d","resolution":{"observed_at":"2026-08-10T23:12:06.793165Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:06.772838Z","title":"UATVR: Uncertainty-Adaptive Text-Video Retrieval,","venue":null,"work_id":"fd1ff323-0ea7-4626-b27e-bc96d337e3ad","year":2023},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:06.022809Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:80ff953cfec073d7d46bf83e351cf396040c7f4bbc39800da66a2222a137734f","observation_id":"133ea5e1-da19-45f1-a8d2-e39daae0b236","resolution":{"observed_at":"2026-08-10T23:12:06.778433Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:06.756862Z","title":"Prompt Switch: Efficient CLIP Adaptation for Text-Video Retrieval,","venue":null,"work_id":"eaf1a3ab-5e1c-4803-bbcd-adbf1011dcc2","year":2023},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:06.027184Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:54661d0ebb01ecaf98adecddda6ec3e3f111ceedc7d0bcf7fed6f9fe53fce9ed","observation_id":"776c9f9a-afb4-46d5-8000-cf06ec459f02","resolution":{"observed_at":"2026-08-10T23:12:06.762238Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:06.741294Z","title":"CenterCLIP: Token Clustering for Efficient Text-Video Retrieval,","venue":null,"work_id":"3f7bf293-78d4-49af-ac4c-fda9629fcc81","year":2022},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:06.031875Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:2d60b39ef2d7fbef92ae57a4efd7736d24f1ef90acc1bd7641e22449f8c70e52","observation_id":"aa1b7248-99e4-48db-94ae-e21fd7fd86ed","resolution":{"observed_at":"2026-08-10T23:12:06.746220Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1807.03748","last_updated":"2019-01-22T18:47:12Z","snapshot_observed_at":"2026-07-06T06:49:24.960992Z","submitted_at":"2018-07-10T16:52:11Z","title":"Representation Learning with Contrastive Predictive Coding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1807.03748","snapshot_observed_at":"2026-08-10T23:12:06.036476Z","title":"Representation Learning with Contrastive Predictive Coding,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:06.036476Z"},"links":{"cited_paper":"/paper/1807.03748","citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:917759495cb63ea0bfb884c4e6dd46e7ff7a5ef6a8b3bcd6faad4240384913ed","observation_id":"da4aa689-5234-4760-b1f2-2ba1ce0e7b18","resolution":{"observed_at":"2026-08-10T23:12:06.036476Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:06.725718Z","title":"Use What You Have: Video Retrieval Using Representations From Collaborative Experts,","venue":null,"work_id":"25f94138-ab8e-4e80-8440-7db660516d53","year":2019},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:06.041258Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:d879f1351c571191ae1654aa3c2895867d0828ed9bb41ac12354d7c295004884","observation_id":"c0fb8e42-135b-425f-b127-0967ae37d3cb","resolution":{"observed_at":"2026-08-10T23:12:06.731010Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:06.710626Z","title":"HowTo100M: Learning a Text-Video Embedding by Watching Hundred Million Narrated Video Clips,","venue":null,"work_id":"694f1d8a-eea8-4033-a455-51661f217cba","year":2019},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:06.045870Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:aaf922e4bc5ac8a2c22cb5095cf65769fbfb6c763705ba9384309462755bf3b8","observation_id":"79c510ef-24e5-44a9-a331-f0b7d7605a6c","resolution":{"observed_at":"2026-08-10T23:12:06.715584Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:06.694219Z","title":"A Joint Sequence Fusion Model for Video Question Answering and Retrieval,","venue":null,"work_id":"8a35fbc7-c0df-4bbf-84ef-abcb73ba34be","year":2018},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:06.050286Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:1fc1d6df8e352e6381eb7febbaad05979436fbec710286c8721c44b9b385e27f","observation_id":"427b624c-1c82-47b8-8a34-34b1837428ce","resolution":{"observed_at":"2026-08-10T23:12:06.700240Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:06.679611Z","title":"BLEU: a Method for Automatic Evaluation of Machine Translation,","venue":null,"work_id":"401e7e65-ea8f-4569-91e9-ce51093d23ad","year":2002},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:06.055538Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:468654d8084ca049a4554ffebe2481ed5a6060d1ed08ef2e3c0e897fd8946abb","observation_id":"b00737e7-0ff0-4a41-a683-c3b4ddc93247","resolution":{"observed_at":"2026-08-10T23:12:06.684401Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2002.06353","last_updated":"2020-09-15T13:27:13Z","snapshot_observed_at":"2026-07-06T08:57:29.493849Z","submitted_at":"2020-02-15T10:03:25Z","title":"UniVL: A Unified Video and Language Pre-Training Model for Multimodal Understanding and Generation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2002.06353","snapshot_observed_at":"2026-08-10T23:12:06.060027Z","title":"UniVL: A Unified Video and Language Pre-Training Model for Multimodal Understanding and Generation,","venue":null,"work_id":null,"year":2002},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:06.060027Z"},"links":{"cited_paper":"/paper/2002.06353","citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:e0ffb71b356e549e4c60ca6eafadfad4d075ba59d24964d300a7b4b88ca59a00","observation_id":"f6780819-c2ea-4a2c-ac1b-75b072106f9e","resolution":{"observed_at":"2026-08-10T23:12:06.060027Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:06.664368Z","title":"METEOR: An Automatic Metric for MT Evaluation with Improved Correlation with Human Judgments,","venue":null,"work_id":"ee257760-2673-4be7-a55a-f90ff94410c1","year":2005},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:06.065222Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:c9f387404afaafed11f81a994e807d1a80ee196c2e22917c289eacd1dbde2dc0","observation_id":"2608b2a6-8f9d-4e4e-aa70-44dc6abe8f24","resolution":{"observed_at":"2026-08-10T23:12:06.669554Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:06.647804Z","title":"CIDEr: Consensus-based Image Description Evaluation,","venue":null,"work_id":"77331abb-aa49-4b85-8566-defe07933525","year":2015},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:06.069878Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:cc5d45928e98574ccd38c2282235a27d34de589edc8f8234733dd748840ff3e5","observation_id":"cc57212a-8e67-4c84-9fdd-f2fe437e7b9f","resolution":{"observed_at":"2026-08-10T23:12:06.653153Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:06.632566Z","title":"Adam: A Method for Stochastic Opti- mization,","venue":null,"work_id":"2b1fd5c0-3332-4f10-a2dc-9c176d2b7eb9","year":2015},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:06.074370Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:97ca14417f41d5f1ff0e55f5a15cdd8478c276e602839d99171df752f7ab8e26","observation_id":"c880074f-2595-4c3a-8a4b-a8fa0607d1bc","resolution":{"observed_at":"2026-08-10T23:12:06.637445Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:06.616940Z","title":"Np-completeness for calculating power IEEE TRANSACTIONS ON PATTERN ANAL YSIS AND MACHINE INTELLIGENCE, VOL. XX,NO. XX, XXX. XXXX 15 indices of weighted majority games,","venue":null,"work_id":"4200324b-a5a9-4312-83d7-aae0381d9727","year":2001},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:06.079203Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:776c541c660fc16bbd287d030858a0effc0e759f60bc0e061c99cd7452e8f337","observation_id":"d62a870f-b1c9-4696-8aba-208a5cf6ce28","resolution":{"observed_at":"2026-08-10T23:12:06.622128Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:06.601965Z","title":"Approximating power indices: theoretical and empirical analysis,","venue":null,"work_id":"950ac1c5-d6a1-41f5-ae27-6bbd525d7177","year":2010},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:06.084651Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:f451d300f382651d1d2af8dd8ae18d8db44e63d301815cd3588cb426f8c17b12","observation_id":"d72011aa-5ba8-428b-93d9-d17126a0a052","resolution":{"observed_at":"2026-08-10T23:12:06.607028Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:06.587499Z","title":"All in One: Exploring Unified Video-Language Pre-training,","venue":null,"work_id":"6a123172-a134-441b-8482-9f0ddbfe7c23","year":2023},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:06.089292Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:95a22e90ff8703ca2254b51360462abcbeb5b42b6b634ed77b373bcd5db36091","observation_id":"8a5e0918-8873-4beb-9cb1-3a340a0a809d","resolution":{"observed_at":"2026-08-10T23:12:06.592171Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:06.573097Z","title":"Zero-Shot Video Question Answering via Frozen Bidirectional Language Models,","venue":null,"work_id":"abc8625d-1676-42d4-8e97-66fc08550831","year":2022},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:06.093859Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:5435b9a9f5eed7bb4f00af8519353c7bfc64ed6aa225b899cbbaa95f7432f537","observation_id":"a4b19c3e-4df0-4fe8-a0b4-93250bc684a1","resolution":{"observed_at":"2026-08-10T23:12:06.577862Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:06.558123Z","title":"Multi-Granularity Interaction and Integration Network for Video Question Answering,","venue":null,"work_id":"7860887e-0c1e-4621-ad0f-256329343f51","year":2023},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:06.098477Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:6333d94364642dcb8f910c6e54ea3cb68b05a6d05bb4dfea66eaa5516b6a4b9d","observation_id":"59788f69-d0ae-421c-b8f0-aad0845698ad","resolution":{"observed_at":"2026-08-10T23:12:06.562922Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:06.542618Z","title":"Invariant Grounding for Video Question Answering,","venue":null,"work_id":"64c1d800-8b3a-4314-8387-7db00a2ec923","year":2022},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:06.103019Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:9a47f20a96cd61a298c20f56396540d648abe23d088659b801b20b93b809cbf5","observation_id":"9dceb0e7-669a-4a6c-9a09-f5cdbcb84564","resolution":{"observed_at":"2026-08-10T23:12:06.547762Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2205.05019","last_updated":"2022-05-11T05:31:08Z","snapshot_observed_at":"2026-07-06T13:08:33.982120Z","submitted_at":"2022-05-10T16:34:26Z","title":"Learning to Answer Visual Questions from Web Videos","version":2},"cited_work":{"arxiv_id":"2205.05019","doi":null,"metadata_source":"pith","pith_arxiv_id":"2205.05019","snapshot_observed_at":"2026-08-10T23:12:06.211553Z","title":"Learning to Answer Visual Questions from Web Videos","venue":"cs.CV","work_id":"cfaefd94-220a-4a8b-ab3d-115a48ca0e78","year":2022},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:06.108321Z"},"links":{"cited_paper":"/paper/2205.05019","citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:7484d42103d519c43d93d1cc4bc0c29d94f9feae6bee63e0f46d3c5e53d84328","observation_id":"03e73832-5573-4e38-a027-897810212dc9","resolution":{"observed_at":"2026-08-10T23:12:06.218997Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:06.527587Z","title":"Video Question Answering With Semantic Disentanglement and Reasoning,","venue":null,"work_id":"9b7944b1-2416-49a1-bb98-98dd5bc835b0","year":2024},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:06.113224Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:6bc34f6f8a190bc590444f04d3067e4a8408f8fab273585cda9634f9442f3ec6","observation_id":"6bbc903e-83ad-4f64-a7e0-0a5b23d8fcc3","resolution":{"observed_at":"2026-08-10T23:12:06.532572Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:06.512050Z","title":"SViTT: Temporal Learning of Sparse Video-Text Transformers,","venue":null,"work_id":"7d6441ff-9970-401f-b623-bd93a87e00bf","year":2023},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:06.117745Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:94226a944ccf3236a7964e6167e863418de5c19898c3c7c2ce455333c9dbbd1b","observation_id":"87fc9f50-ec22-48ff-8463-251123d092c2","resolution":{"observed_at":"2026-08-10T23:12:06.516916Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:06.496856Z","title":"TG-VQA: Ternary Game of Video Question Answering,","venue":null,"work_id":"e7103986-f891-47e6-a4cd-398ef2e8f368","year":2023},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:06.122401Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:b4d228631f6b66ecde75198859484e5656c2a8d421b6fed17a7e587d4db80fd9","observation_id":"b1503c78-c126-43f8-8679-bb33c5a59b99","resolution":{"observed_at":"2026-08-10T23:12:06.501948Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:06.481710Z","title":"SWINBERT: End-to-End Transformers with Sparse Attention for Video Captioning,","venue":null,"work_id":"4b20eb57-ad60-43f0-a183-c5aeb863002d","year":2022},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:06.127080Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:439b306dd22e83975d71b13de6193279f3152cad1bb222468b4cc62afc31de05","observation_id":"eef74904-de05-42e9-a437-f6a71b29b991","resolution":{"observed_at":"2026-08-10T23:12:06.486822Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:06.465285Z","title":"End-to-end Generative Pretraining for Multimodal Video Captioning,","venue":null,"work_id":"8dd802fd-caa8-49ea-a3a5-4076a5c1aab6","year":2022},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:06.131507Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:23e2d62e0f67ca16d5477f2c3e4876d413fcf7d7f1dc52c6fd174f29e5188365","observation_id":"0a4ac961-06e3-4e5a-b2be-fe544deeec82","resolution":{"observed_at":"2026-08-10T23:12:06.471147Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:06.449875Z","title":"Motion Guided Region Message Passing for Video Captioning,","venue":null,"work_id":"39c19ffa-8516-432d-899a-0fcfc504af7c","year":2021},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:06.136069Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:b097a0838d51c3197624a51793395c4ca115fa535ab4ab9136b87c249e4e7bcd","observation_id":"e5f9d9c6-66ce-4ac3-b6ef-64bcf0c1f83a","resolution":{"observed_at":"2026-08-10T23:12:06.454914Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:06.434917Z","title":"Open-book Video Captioning with Retrieve-Copy-Generate Net- work,","venue":null,"work_id":"a403275e-841a-4e13-bc19-bc17f7ca2652","year":2021},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:06.140645Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:70ffea1179d1e0724f943f0ec15a8885f16055326cd2c08f40774cd59dcb62d6","observation_id":"49ad36ec-6710-4eed-85f8-b9f6df6b9bd7","resolution":{"observed_at":"2026-08-10T23:12:06.439692Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:06.419576Z","title":"Attentive Visual Semantic Specialized Network for Video Captioning,","venue":null,"work_id":"db9fd272-0913-482f-8728-81656aef5ef3","year":2021},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:06.145153Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:04a13caa598a5607e7fe369629a14aa21fd43a2740e052980edfcfee1fc8bbe0","observation_id":"7cffd5d3-dc94-43ff-802b-23f791be790c","resolution":{"observed_at":"2026-08-10T23:12:06.424464Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:06.403116Z","title":"Global semantic enhancement network for video captioning,","venue":null,"work_id":"de99ac82-0994-4bc8-ac22-e6edf40fc1c3","year":2024},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":90,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:06.149625Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:571eb5debbb26dc8dc1c9ebacff99bfc9cae0c007137fe2200cd25db3394cf54","observation_id":"6bcd28a6-c000-49e9-98c4-a62a84bc9b9c","resolution":{"observed_at":"2026-08-10T23:12:06.408778Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:06.386563Z","title":"Accurate and Fast Compressed Video Captioning,","venue":null,"work_id":"fa547ef2-e825-48b2-94fa-1b0977ea0b2c","year":2023},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":91,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:06.153955Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:5f7cded2abd49318fc06b32e2e77e1b634868cbc27bb9ed028f3cc5739a08463","observation_id":"5b245349-1528-467a-a873-cb780712c22e","resolution":{"observed_at":"2026-08-10T23:12:06.391481Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:06.370849Z","title":"Emotional Video Captioning with Vision-based Emotion Interpretation Network,","venue":null,"work_id":"719f55c5-0258-4b0c-8cf5-c3cfcc1977af","year":2024},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:06.158851Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:79654cf24836d298d07c1d4800e0a06c95f6144ca6d9b8768a84241d38dba4f4","observation_id":"5356a222-97a4-4788-8cd2-56edf440d795","resolution":{"observed_at":"2026-08-10T23:12:06.375787Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:06.355683Z","title":"Improving Video Cap- tioning with Temporal Composition of a Visual-Syntactic Em- bedding,","venue":null,"work_id":"3dd902c1-7b40-4e96-97f1-8b77dff6b66d","year":2021},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":93,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:06.163381Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:d6e4ba626c6cf6a5549a6048a0ec432dbd13853c5b025a676d061e195058301a","observation_id":"989e1414-4faa-433e-81f7-9d43588b5c27","resolution":{"observed_at":"2026-08-10T23:12:06.360802Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:06.340139Z","title":"CLIP4Caption: CLIP for Video Caption,","venue":null,"work_id":"57b0b3a7-724d-4922-92b6-a0f3eda6a84f","year":2021},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:06.167875Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:f91a581dbb60712596643785d960f3f4ff1b52ebf956af832d119583434b8538","observation_id":"2b4da795-3272-4e61-94e2-545790861b51","resolution":{"observed_at":"2026-08-10T23:12:06.345385Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:12:06.325161Z","title":"Visualizing data using t-SNE,","venue":null,"work_id":"9914adcb-a2ea-4a1e-afe9-5015224b2651","year":2008},"citing_paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning","version":1},"reference_index":95,"source":"pdf_text","source_observed_at":"2026-08-10T23:12:06.172447Z"},"links":{"citing_paper":"/paper/2412.20964"},"observation_digest":"sha256:a39fcbbe9253f486b51b4ff90b9a34fd04b00b4c18f95099efcc29b20dbd2e09","observation_id":"44183826-f439-4dc4-baa9-53623641cdf6","resolution":{"observed_at":"2026-08-10T23:12:06.330034Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2412.20964","last_updated":"2024-12-30T14:09:15Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-12T02:07:07.949624Z","submitted_at":"2024-12-30T14:09:15Z","title":"Hierarchical Banzhaf Interaction for General Video-Language Representation Learning"},"reference_resolution":{"displayed":95,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":23,"verified_exact":1,"verified_fuzzy":71},"total_outbound_references":95},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"thesis":"As of 13 August 2026, this Paper Citation Record lists 95 of 95 outbound references and 0 inbound Pith citation observations for arXiv:2412.20964."}