{"as_of":"2026-08-14T09:23:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:a8ce6079244b400ccc7ee6752ce3c1edddeaae15d1a5c6343f13fd451903676c","coverage":[{"denominator":132,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T21:27:16.884770Z","state":"measured"},{"denominator":102,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":102,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-14T06:32:32.682623+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-11T04:45:38.441181Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-07T06:00:58.843605Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.08753","snapshot_observed_at":"2026-08-11T04:45:38.441181Z","title":"Which viewpoint shows it best? language for weakly supervising view selection in multi- view videos","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.18386","last_updated":"2025-04-22T13:23:34Z","snapshot_observed_at":"2026-08-12T23:13:41.357483Z","submitted_at":"2024-12-24T12:16:43Z","title":"Switch-a-View: View Selection Learned from Unlabeled In-the-wild Videos","version":3},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-11T04:45:38.441181Z"},"links":{"cited_paper":"/paper/2411.08753","citing_paper":"/paper/2412.18386"},"observation_digest":"sha256:98420a05001f27b9abc4d4cd09251b83894218d0a45b059c9f2c475eb2ae3601","observation_id":"6cd707ae-c6d6-4b32-b3a1-eec071956336","resolution":{"observed_at":"2026-08-11T04:45:38.441181Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"cited_work":{"arxiv_id":"2411.08753","doi":null,"metadata_source":"pith","pith_arxiv_id":"2411.08753","snapshot_observed_at":"2026-08-07T06:00:58.843605Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","venue":"cs.CV","work_id":"ff735e17-5ea6-4d37-ba76-90bb159da45b","year":2024},"citing_paper":{"arxiv_id":"2506.06253","last_updated":"2025-06-06T17:25:48Z","snapshot_observed_at":"2026-08-09T07:19:17.693738Z","submitted_at":"2025-06-06T17:25:48Z","title":"Bridging Perspectives: A Survey on Cross-view Collaborative Intelligence with Egocentric-Exocentric Vision","version":1},"reference_index":187,"source":"pdf_text","source_observed_at":"2026-08-07T06:00:58.555825Z"},"links":{"cited_paper":"/paper/2411.08753","citing_paper":"/paper/2506.06253"},"observation_digest":"sha256:1328d2c2736c00901bb1fd7be866358e8959e5b5246654079d26acbb6cefaa81","observation_id":"ebe94ece-1e15-49a0-bcc6-e24f2cfb1f29","resolution":{"observed_at":"2026-08-07T06:00:58.847019Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2411.08753/citation-record","integrity":"/paper/2411.08753/integrity","json":"/paper/2411.08753/citation-record.json","paper":"/paper/2411.08753"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"1803.08375","last_updated":"2026-04-14T12:21:53Z","snapshot_observed_at":"2026-08-14T01:11:07.622125Z","submitted_at":"2018-03-22T14:30:17Z","title":"Deep Learning using Rectified Linear Units (ReLU)","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1803.08375","snapshot_observed_at":"2026-08-12T21:27:16.239636Z","title":"Deep learning using rectified linear units (relu), 2018","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.239636Z"},"links":{"cited_paper":"/paper/1803.08375","citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:3a46220e34570c75ebe29bd4e644c3886162bfe911b816b1cc8679d2553bf229","observation_id":"bb025d11-7abc-45fa-941e-be55fa65712e","resolution":{"observed_at":"2026-08-12T21:27:16.239636Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.246965Z","title":"McCrae, Kenton Murray, Maria Nadejde, Satoshi Nakamura, Matteo Negri, Ha Nguyen, Jan Niehues, Xing Niu, Atul Kr","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.246965Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:89313189c9ed9bfea206c9920b383c242c5e76bb14569dadd1b338afd876e3ff","observation_id":"bbbd5838-3012-4a00-bed1-68e5f1a12eb2","resolution":{"observed_at":"2026-08-12T21:27:16.246965Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.253155Z","title":"A dataset for develop- ing and benchmarking active vision","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.253155Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:33213fb86b1cc291c5fb39d58e415da1699ff9b9e4472b572a1a8c69505487b8","observation_id":"5f7e63c4-d784-42c4-9cd6-80c062bfff39","resolution":{"observed_at":"2026-08-12T21:27:16.253155Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.258665Z","title":"Automatic editing of footage from multi- ple social cameras","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.258665Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:9082c2260ff4d9dc380baf95b7c47a13ac1cf63bb482193b8a0c915723de5ff1","observation_id":"5f35b769-c0b0-401e-b12b-d158305d7e1d","resolution":{"observed_at":"2026-08-12T21:27:16.258665Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.266710Z","title":null,"venue":null,"work_id":null,"year":1988},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.266710Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:63a13581bcac91eb40008eeeac04c3ba671502334a1636695a6787e39b658a23","observation_id":"2b809dd6-454e-4425-a264-bae399839d18","resolution":{"observed_at":"2026-08-12T21:27:16.266710Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2012.02356","last_updated":"2021-05-28T07:09:48Z","snapshot_observed_at":"2026-08-13T23:53:49.363179Z","submitted_at":"2020-12-04T01:22:05Z","title":"WeaQA: Weak Supervision via Captions for Visual Question Answering","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2012.02356","snapshot_observed_at":"2026-08-12T21:27:16.273850Z","title":"Weaqa: Weak supervision via captions for visual question answering","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.273850Z"},"links":{"cited_paper":"/paper/2012.02356","citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:573de93743d75c3060aa6ba4d5e0f849f794b5ec445d4e031998a8783cb354a1","observation_id":"cc8180e4-325e-4e9e-8930-f4a1fede591c","resolution":{"observed_at":"2026-08-12T21:27:16.273850Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.281097Z","title":"METEOR: An auto- matic metric for MT evaluation with improved correlation with human judgments","venue":null,"work_id":null,"year":2005},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.281097Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:5c1ae5285983579ebbbaac025919a85533fbabefb06e06bdc61b8f1d084f8025","observation_id":"a3137ed3-695d-457c-b997-5a27ba7b661f","resolution":{"observed_at":"2026-08-12T21:27:16.281097Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.287219Z","title":"Is space-time attention all you need for video understanding? In ICML, page 4, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.287219Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:04a52d274544ff6e651b5681f7d6fceaa3613630d5349b6d3b90ba157c924ed1","observation_id":"1fd79043-4e34-4a2c-892b-dd4ab3137967","resolution":{"observed_at":"2026-08-12T21:27:16.287219Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.296787Z","title":"High- lightme: Detecting highlights from human-centric videos","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.296787Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:0580fa9a6d74948fe9b7cf094cf578bde2ffe3d21e502a90269cb6cbab8c6e18","observation_id":"71bc575b-f19d-4b46-8a5c-939063e2b0b4","resolution":{"observed_at":"2026-08-12T21:27:16.296787Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.304503Z","title":"Extreme rotation estimation using dense correlation volumes","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.304503Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:a0e819cb9897fbdc4bef4ac3e1b60b845665991e5902386744ac8bd6cfbd1b4a","observation_id":"1970d812-711b-4c45-a19f-73c8b38c4a28","resolution":{"observed_at":"2026-08-12T21:27:16.304503Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.312059Z","title":"Davis, and Lei Zhang","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.312059Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:750f2810cd3bf0493ea17233dd49a3cac7a174fd689c267b77197a0a88e751f1","observation_id":"0a0c9b5d-fb6a-4dba-a770-a30c788473ae","resolution":{"observed_at":"2026-08-12T21:27:16.312059Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.318489Z","title":"Enhanced interactive 360° viewing via automatic guidance","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.318489Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:743df688973e01432d53b914519dffecb6d51cf002e8879a6f190390465a21ef","observation_id":"e8215ca4-bb01-45c4-917f-b677ab0198c0","resolution":{"observed_at":"2026-08-12T21:27:16.318489Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.324741Z","title":"Learn- ing sports camera selection from internet videos","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.324741Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:fbde3b5e09a2c2cea6125448f9ea3d557bfa7d5c0406655bbeac1c4f0a186a89","observation_id":"20ccd794-4833-45b6-95bf-6553e0024242","resolution":{"observed_at":"2026-08-12T21:27:16.324741Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.331782Z","title":"Wide- baseline relative camera pose estimation with directional learning","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.331782Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:a490c2a93f5bdd9f38320f9f05b335d112ebfb7eeb94bf3d1a8c20c14b65f081","observation_id":"9f637a90-1a36-4165-9a78-1202d46acbd4","resolution":{"observed_at":"2026-08-12T21:27:16.331782Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.337905Z","title":"Geometry-aware recurrent neural networks for active visual recognition","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.337905Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:eb5e5fbbe4a3644d339ccf41835a68b95e847919643866265f66cc2c40a26929","observation_id":"3899a749-9f6c-48a5-bf29-582af09d9ade","resolution":{"observed_at":"2026-08-12T21:27:16.337905Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.351292Z","title":"Towards a richer 2d understanding of hands at scale","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.351292Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:103b7302d739283b00515943bf38c8376e08cb0c32ebdb578550da087c24c323","observation_id":"128e8bf9-6538-4aff-8651-901cd8535971","resolution":{"observed_at":"2026-08-12T21:27:16.351292Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.357941Z","title":"Gonzalez, Ion Stoica, and Eric P","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.357941Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:aae06d3bfd69f99019078b7931da5735b7da3711d55f80d8f8b2f145d9d0a690","observation_id":"f512be94-a5f9-47ad-9c48-3fe5ebc5064b","resolution":{"observed_at":"2026-08-12T21:27:16.357941Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1711.08664","last_updated":"2017-11-23T12:06:20Z","snapshot_observed_at":"2026-08-04T01:34:38.975772Z","submitted_at":"2017-11-23T12:06:20Z","title":"Self-view Grounding Given a Narrated 360{\\deg} Video","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1711.08664","snapshot_observed_at":"2026-08-12T21:27:16.365365Z","title":"Self-view ground- ing given a narrated 360 {\\deg} video","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.365365Z"},"links":{"cited_paper":"/paper/1711.08664","citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:33e5d0e020b066394f67c0181a3eba4ae9db062f1b314291f66c8177c309e6cb","observation_id":"85501816-c5e0-4811-a324-9150f4030214","resolution":{"observed_at":"2026-08-12T21:27:16.365365Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.374896Z","title":"Video co-summarization: Video summarization by visual co- occurrence","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.374896Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:79fc0c777b074d7c1947ce72017f34413987c76aeb552c4343d5d19f6bf56db3","observation_id":"0b7dded2-0479-4e31-860c-2989caae0cf8","resolution":{"observed_at":"2026-08-12T21:27:16.374896Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.381981Z","title":"elochoice","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.381981Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:8f402bf52bb92835fe0f88048d0eb0e36887abfdf79333e53a7447b60e5a2433","observation_id":"7ab7f5a7-5e64-401d-b3e0-1b9d56e9b9a2","resolution":{"observed_at":"2026-08-12T21:27:16.381981Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.387951Z","title":"Scaling egocentric vision: The epic- kitchens dataset","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.387951Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:d4a0c64b615cee64c79175abc306407ec826f7d9d707563619228529e7d1ab2f","observation_id":"15cdcf51-af90-48d1-9276-bccd1ddac3ad","resolution":{"observed_at":"2026-08-12T21:27:16.387951Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.08691","last_updated":"2023-07-17T17:50:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-17T17:50:36Z","title":"FlashAttention-2: Faster Attention with Better Parallelism and Work Partitioning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.08691","snapshot_observed_at":"2026-08-12T21:27:16.393588Z","title":"Flashattention-2: Faster attention with bet- ter parallelism and work partitioning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.393588Z"},"links":{"cited_paper":"/paper/2307.08691","citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:0b102a99205b69b794cee7c7f3f491b9d61619f678a794fc47fdd2707608dc8a","observation_id":"13f61efd-951b-41cd-a55b-661009310c20","resolution":{"observed_at":"2026-08-12T21:27:16.393588Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.399483Z","title":"Flashattention: Fast and memory-efficient exact attention with io-awareness","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.399483Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:bc99ef5ce0c5b1ac559238e1073da45ee7327854e5b31cc6a771d9c3f2ada833","observation_id":"58020f30-05f1-4628-8f05-3ada0da5fc75","resolution":{"observed_at":"2026-08-12T21:27:16.399483Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.404733Z","title":"Velastin","venue":null,"work_id":null,"year":2008},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.404733Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:6016dbeb962e291391b80dd9053e486917065d86400b5daae441c6641340e209","observation_id":"9cb4724f-b0dd-49f3-825f-fe2f18162725","resolution":{"observed_at":"2026-08-12T21:27:16.404733Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.410436Z","title":"Virtex: Learning visual representations from textual annotations","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.410436Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:45c850476e53d1f5608b129aa5496f7c97f17615686c901e201b07a159e9c7bf","observation_id":"d0be0b1f-e948-4cb3-b0af-4b86378f439e","resolution":{"observed_at":"2026-08-12T21:27:16.410436Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2010.11929","last_updated":"2021-06-03T13:08:56Z","snapshot_observed_at":"2026-08-13T14:19:26.598265Z","submitted_at":"2020-10-22T17:55:59Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.11929","snapshot_observed_at":"2026-08-12T21:27:16.416234Z","title":"An image is worth 16x16 words: Trans- formers for image recognition at scale","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.416234Z"},"links":{"cited_paper":"/paper/2010.11929","citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:0467d0637381af2e3885a2e22c469e12318a9259d401cafcd7f9945b5c07949c","observation_id":"acde62f6-cf94-455a-bc65-539a3d3f779d","resolution":{"observed_at":"2026-08-12T21:27:16.416234Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.423479Z","title":"Dense and aligned captions (dac) promote compositional reasoning in vl models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.423479Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:0abfd733f1a085e46f7d39197504c7ebf297708d34fe40b6ed59baf27f3151d0","observation_id":"786245bb-34cc-4782-b6da-f8e960b5843f","resolution":{"observed_at":"2026-08-12T21:27:16.423479Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.429622Z","title":"Multi-view active fine- grained visual recognition","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.429622Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:ed56242d98b30738109d7dcf1e1c481e41221a0c692c7d8552d92742f70b47d0","observation_id":"face0574-4d45-4a9a-b547-52f30f5a2cb4","resolution":{"observed_at":"2026-08-12T21:27:16.429622Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.436093Z","title":"Multi- stream dynamic video summarization","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.436093Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:317771f7bd9fb31e5fb5e16b99c9e8ffbf2ac8c7b2155ad4f2e4528d79dcbe5f","observation_id":"ae7829db-5a61-4529-b7b2-53d081487132","resolution":{"observed_at":"2026-08-12T21:27:16.436093Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.441602Z","title":"Elson and Mark O","venue":null,"work_id":null,"year":2007},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.441602Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:1eaa30cff752d225be9577da51c87e312436ab1ac2b1a2855d6c671e90865a2c","observation_id":"6c41b363-bd5c-48fc-8940-827deb8740d4","resolution":{"observed_at":"2026-08-12T21:27:16.441602Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.450299Z","title":"Foote and D","venue":null,"work_id":null,"year":2000},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.450299Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:0a4d9d0e14abf2a0ed39a6eb917d23b5bf22655a931478abe0b7bfa20852e06f","observation_id":"20a10721-ce03-4f24-8f5d-68767cb25750","resolution":{"observed_at":"2026-08-12T21:27:16.450299Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.457120Z","title":"Multi-view video summa- rization","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.457120Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:3091195e366652f84ee83d6ec71c22ec0a50a14c0ffe6453338973ca06369b22","observation_id":"0904ea70-ee80-4a76-9aac-ec3227472dea","resolution":{"observed_at":"2026-08-12T21:27:16.457120Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.463492Z","title":"Gleicher, Rachel M","venue":null,"work_id":null,"year":2002},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.463492Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:4af44867cd4ca604699bfaee9b28922319b30c0c614ccd252e1a502550de8a9b","observation_id":"aea03a6d-8bf6-41c2-823e-ae6aaa5262d3","resolution":{"observed_at":"2026-08-12T21:27:16.463492Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.07336","last_updated":"2024-04-10T20:32:24Z","snapshot_observed_at":"2026-08-13T00:33:06.461321Z","submitted_at":"2024-04-10T20:32:24Z","title":"PEAVS: Perceptual Evaluation of Audio-Visual Synchrony Grounded in Viewers' Opinion Scores","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.07336","snapshot_observed_at":"2026-08-12T21:27:16.469572Z","title":"Peavs: Perceptual evaluation of audio-visual syn- chrony grounded in viewers’ opinion scores","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.469572Z"},"links":{"cited_paper":"/paper/2404.07336","citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:efe84c454bc10303c825159816eae78da2f6bf20a9e4696d97def49a34507a56","observation_id":"69cd9eef-ad25-4bde-be6c-0f56da3ca3fa","resolution":{"observed_at":"2026-08-12T21:27:16.469572Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.479571Z","title":"Diverse sequential subset selection for supervised video summarization","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.479571Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:ff2ed3cf619959b0b012175588212c42b742ac2c95da9b7ce186d2c69954b8c0","observation_id":"b399e38c-c960-45a5-8172-81faef191d2f","resolution":{"observed_at":"2026-08-12T21:27:16.479571Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.485303Z","title":"Ego4d: Around the world in 3,000 hours of egocentric video","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.485303Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:aab6d274001233a4c752ab7058c5b854e765b9e42a4eb7b5a8d38183477e0970","observation_id":"1fb9a401-425c-4436-beaa-101ae1ca7ace","resolution":{"observed_at":"2026-08-12T21:27:16.485303Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.18259","last_updated":"2024-09-25T21:55:38Z","snapshot_observed_at":"2026-08-13T05:13:48.123305Z","submitted_at":"2023-11-30T05:21:07Z","title":"Ego-Exo4D: Understanding Skilled Human Activity from First- and Third-Person Perspectives","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.18259","snapshot_observed_at":"2026-08-12T21:27:16.490447Z","title":"Ego-exo4d: Understanding skilled human activity from first-and third-person perspectives","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.490447Z"},"links":{"cited_paper":"/paper/2311.18259","citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:a0fb9728fe89fc2ae3f42e6dc29bc76cf4a8076077846aa3d71c7060d90fdd37","observation_id":"c0b2dc9a-8d8c-43ed-9738-c519d6a0c022","resolution":{"observed_at":"2026-08-12T21:27:16.490447Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1806.03107","last_updated":"2019-01-02T16:53:13Z","snapshot_observed_at":"2026-08-13T21:52:58.954758Z","submitted_at":"2018-06-08T12:10:58Z","title":"Temporal Difference Variational Auto-Encoder","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1806.03107","snapshot_observed_at":"2026-08-12T21:27:16.496223Z","title":"Temporal difference varia- tional auto-encoder","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.496223Z"},"links":{"cited_paper":"/paper/1806.03107","citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:e977d37616fc7c82b9bc43ca2606ffa503f4a38f3d24ae3a7c55ebf24719e6d4","observation_id":"36ada9b3-36b4-4a4b-bdaf-73478b66a597","resolution":{"observed_at":"2026-08-12T21:27:16.496223Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2212.10846","last_updated":"2023-05-08T06:04:04Z","snapshot_observed_at":"2026-08-13T13:17:30.274656Z","submitted_at":"2022-12-21T08:39:36Z","title":"From Images to Textual Prompts: Zero-shot VQA with Frozen Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.10846","snapshot_observed_at":"2026-08-12T21:27:16.503459Z","title":"From images to textual prompts: Zero-shot vqa with frozen large language models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.503459Z"},"links":{"cited_paper":"/paper/2212.10846","citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:edd4c819bacf7e30306e5cb54f424779bfb1cca3ca961c403eed6066bc2bb535","observation_id":"82cf2f02-60ce-48a7-b2b7-45c7745d63b4","resolution":{"observed_at":"2026-08-12T21:27:16.503459Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.510211Z","title":"Using closed captions as supervision for video activity recognition","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.510211Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:12196bdea35b8c6d2113d51abec4d5e0e95fa10cdf02bc0738d31e741eaf5ff6","observation_id":"bb7f5658-868b-4da4-8003-6fea52ca632f","resolution":{"observed_at":"2026-08-12T21:27:16.510211Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.520424Z","title":"Creating summaries from user videos","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.520424Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:8e62423b19d440ff44c0c060d9d6181cc0bc026fd39b9b2d5cd6a57c6c632818","observation_id":"ee3b7321-6456-4c50-a586-d310a09ced60","resolution":{"observed_at":"2026-08-12T21:27:16.520424Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.533989Z","title":"Video summarization by learning submodular mixtures of objec- tives","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.533989Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:08a56c350ed952091dbc807e263aa39b85148aceac91de8776d79d71cefec6ca","observation_id":"adf155bf-20d6-4c5c-b6cc-30895c8ea6ba","resolution":{"observed_at":"2026-08-12T21:27:16.533989Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.539969Z","title":"Align and attend: Multimodal summarization with dual contrastive losses","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.539969Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:05a1c767366cfffd75566a18bf59cb9dcb7f785d77776ab01c4b923676799aa0","observation_id":"51c0c39d-eb07-4531-b1e0-32883fb4dd9b","resolution":{"observed_at":"2026-08-12T21:27:16.539969Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.545269Z","title":"Cohen, and David H","venue":null,"work_id":null,"year":1996},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.545269Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:33772ed4504ddcc8ae8ea3288afc17ebd1f69caa4aa3981aa4cef2d808f29617","observation_id":"ccfe8115-fc24-4cb6-8957-db3605f10956","resolution":{"observed_at":"2026-08-12T21:27:16.545269Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.551100Z","title":"Cohen, and David H","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.551100Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:3f17dec6c0a37b0f2f6f894bb06a92acaa09b4ff9bf901c09074888ffdaa1037","observation_id":"4771a9c2-2ac3-4484-be01-c8bc7857d19c","resolution":{"observed_at":"2026-08-12T21:27:16.551100Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.556351Z","title":"Vir- tual videography","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.556351Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:863d861754d78efadabfeb670f9e59e10e248e6fe0a3366d62e4f572f39a9482","observation_id":"772d4bde-00fa-4203-a507-cb95825b6041","resolution":{"observed_at":"2026-08-12T21:27:16.556351Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.09685","last_updated":"2021-10-16T18:40:34Z","snapshot_observed_at":"2026-08-11T08:20:29.798517Z","submitted_at":"2021-06-17T17:37:18Z","title":"LoRA: Low-Rank Adaptation of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.09685","snapshot_observed_at":"2026-08-12T21:27:16.561315Z","title":"Lora: Low-rank adaptation of large language models","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.561315Z"},"links":{"cited_paper":"/paper/2106.09685","citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:61b7e55858f16958bfe69fa13706548776576a473a3620986757a6cf0b0e978a","observation_id":"00f95f17-6312-4f34-91fb-bd507296c3bd","resolution":{"observed_at":"2026-08-12T21:27:16.561315Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.566908Z","title":"Deep 360 pilot: Learning a deep agent for piloting through 360deg sports videos","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.566908Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:67be56a9e836936bb0663f787462e86d415ad10eaa4bc2be0a00c1bf61d782a5","observation_id":"d38383b6-f858-4707-a017-1dede39c461d","resolution":{"observed_at":"2026-08-12T21:27:16.566908Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.16182","last_updated":"2025-03-06T02:46:51Z","snapshot_observed_at":"2026-08-13T00:46:57.055849Z","submitted_at":"2024-03-24T15:00:44Z","title":"EgoExoLearn: A Dataset for Bridging Asynchronous Ego- and Exo-centric View of Procedural Activities in Real World","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.16182","snapshot_observed_at":"2026-08-12T21:27:16.572054Z","title":"Egoexolearn: A dataset for bridging asynchronous ego-and exo-centric view of procedural activi- ties in real world","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.572054Z"},"links":{"cited_paper":"/paper/2403.16182","citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:1f6e10383071e4cc8bbbd95451138f955847dcd247cf077f26e9b4fdd5dc3a25","observation_id":"eaaccbea-74bd-4aa4-a9a1-0ed54ce64a55","resolution":{"observed_at":"2026-08-12T21:27:16.572054Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.577856Z","title":"Batch normalization: accelerating deep network training by reducing internal co- variate shift","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.577856Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:cc889cdfb43f3df47035b92dfe7b0b876f4a13abe9b621fb851ee736b5fe5dd8","observation_id":"d3df167d-f2ce-44a1-97d9-fcefb64f1699","resolution":{"observed_at":"2026-08-12T21:27:16.577856Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.584139Z","title":"Look-ahead be- fore you leap: end-to-end active recognition by forecasting the effect of motion","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.584139Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:ff852bcf24178dac3fd5dae83082ea3530692c1e1060bd02e580a5aa9436f841","observation_id":"c0397a64-a4a6-40dc-9e1d-f306b0a6255a","resolution":{"observed_at":"2026-08-12T21:27:16.584139Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.589659Z","title":"Learning to look around: Intelligently exploring unseen environments for unknown tasks","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.589659Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:3a858d14acdf8450a350cfb52516fdc812377c5b6f3ee561b7629d507f929631","observation_id":"8ea4e95b-7769-44eb-b8c4-b79cb4233be3","resolution":{"observed_at":"2026-08-12T21:27:16.589659Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.595297Z","title":"End-to-end policy learning for active visual categorization","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.595297Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:95d29ee5976c138e3a7b9516418dab377751efa1f16c028a146f3137d73076fe","observation_id":"512bb144-e0c5-4bf9-a5f8-977456cfdf53","resolution":{"observed_at":"2026-08-12T21:27:16.595297Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1808.07784","last_updated":"2018-10-23T19:22:25Z","snapshot_observed_at":"2026-07-06T06:57:04.785555Z","submitted_at":"2018-08-23T14:52:40Z","title":"Time-Agnostic Prediction: Predicting Predictable Video Frames","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1808.07784","snapshot_observed_at":"2026-08-12T21:27:16.600840Z","title":"Time-agnostic prediction: Predicting pre- dictable video frames","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.600840Z"},"links":{"cited_paper":"/paper/1808.07784","citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:349d8e2177e858bc3b331372da472266b516a7231e5dc573d8a29153ca412881","observation_id":"2835bd81-9f36-41e4-a29d-cbf613bc3efc","resolution":{"observed_at":"2026-08-12T21:27:16.600840Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.607025Z","title":"Simglim: Simplifying glimpse based active visual reconstruction","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.607025Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:968751f5ccdeea0e180ab7fb3fa027e8abd6d6ccdfc4470ff2b585859268b1c9","observation_id":"5c8021b6-ddd4-4df0-a69b-51c0e2d0c7ce","resolution":{"observed_at":"2026-08-12T21:27:16.607025Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.612958Z","title":"Lemma: A multi-view dataset for le arning m ulti-agent m ulti-task a ctivities","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.612958Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:0fbf0a5d4f5afa87de7ee506d0f7ebb44e01c3a1ea2fed11d1d9ff85363c3d1b","observation_id":"d3adfba1-3679-4606-90ce-c0a8591e3d1d","resolution":{"observed_at":"2026-08-12T21:27:16.612958Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.07399","last_updated":"2023-07-03T03:06:26Z","snapshot_observed_at":"2026-08-13T12:25:27.314050Z","submitted_at":"2023-03-13T18:26:11Z","title":"RTMPose: Real-Time Multi-Person Pose Estimation based on MMPose","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.07399","snapshot_observed_at":"2026-08-12T21:27:16.619160Z","title":"Rtmpose: Real-time multi-person pose estimation based on mmpose","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.619160Z"},"links":{"cited_paper":"/paper/2303.07399","citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:fba267c136284788e10cfdff54d08c3a6f78fde077b5915cde2ed5cf2c11d374","observation_id":"75304319-c3b5-4630-854f-269cd0168e43","resolution":{"observed_at":"2026-08-12T21:27:16.619160Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.624405Z","title":"Large-scale video summarization using web-image priors","venue":null,"work_id":null,"year":2013},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.624405Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:ea1a4eda26266d422e7c12ce0caef27163276fffe49f5097e3afb7564d3c2c82","observation_id":"cece5df7-2619-49f6-be83-94a62c0fa7f0","resolution":{"observed_at":"2026-08-12T21:27:16.624405Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.629375Z","title":null,"venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.629375Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:b24f9e976534ee9b5ca552b58546f6f20bd2dbbe5654a22220111db4c3c3fc1f","observation_id":"4afe155d-955c-48ed-bbcc-329d56768cd9","resolution":{"observed_at":"2026-08-12T21:27:16.629375Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.635697Z","title":"Segment any- thing","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.635697Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:9ded63a5c23f8703a93d16b6f9425fdf206591d0c8d12b04e4d13c228c5946be","observation_id":"cb661ab1-ee16-4e98-81ab-c87070f9aa7b","resolution":{"observed_at":"2026-08-12T21:27:16.635697Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.05016","last_updated":"2024-04-07T17:06:22Z","snapshot_observed_at":"2026-08-13T00:35:31.505586Z","submitted_at":"2024-04-07T17:06:22Z","title":"Hyperbolic Learning with Synthetic Captions for Open-World Detection","version":1},"cited_work":{"arxiv_id":"2404.05016","doi":null,"metadata_source":"pith","pith_arxiv_id":"2404.05016","snapshot_observed_at":"2026-08-12T21:27:17.283091Z","title":"Hyperbolic Learning with Synthetic Captions for Open-World Detection","venue":"cs.CV","work_id":"e1ef6378-7af6-4bfa-be0b-d4ca7da54b39","year":2024},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.641426Z"},"links":{"cited_paper":"/paper/2404.05016","citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:328af4268e15a3bf0e5ab6896dc5bcd5058ed742b502eb4b4c411a0411453961","observation_id":"883c89c2-d078-4ec9-9dee-f367fe283eba","resolution":{"observed_at":"2026-08-12T21:27:17.290381Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.650230Z","title":"A memory network approach for story-based temporal summarization of 360° videos","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.650230Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:1bce3d24348f18a4a5a6f5a48f3adbcd4f2c57a57df1d4cb8fb2d18af2a74a08","observation_id":"41dfb3fc-a295-4cc4-895d-81fc3c3afa56","resolution":{"observed_at":"2026-08-12T21:27:16.650230Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.657826Z","title":"Predicting important objects for egocentric video summarization","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.657826Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:8975c92561513a104cf62fdfecc66c6e6f5d6a0e0c84190ae12c98e8c677892e","observation_id":"f6e45f0f-680a-4afb-9c6e-ea8b2a8a707a","resolution":{"observed_at":"2026-08-12T21:27:16.657826Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-08-12T21:27:16.663653Z","title":"Mvbench: A comprehensive multi-modal video understand- ing benchmark","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.663653Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:49e864988a6c85c21b5495e82927fc039da04b658a4d11811647c1eefe1c4831","observation_id":"ebfefeba-e6cc-4e98-ad04-5e03f150fd7a","resolution":{"observed_at":"2026-08-12T21:27:16.663653Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.669888Z","title":"Grounded language-image pre-training","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.669888Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:ed4109587d539d4c6ea45cc8bb636a2b1356b02e6897bcd64448a819e21facf0","observation_id":"e15fd5d7-482f-47e8-9912-453c1c504e1c","resolution":{"observed_at":"2026-08-12T21:27:16.669888Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.675226Z","title":"How local is the local diversity? reinforcing sequen- tial determinantal point processes with dynamic ground sets for supervised video summarization","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.675226Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:6483535d82a356b3918402757a0b6fff12d95bb143c9dd90ddbb316dc6cbc2ca","observation_id":"cb28cbc6-1078-4f06-883f-5b54aacfb4cb","resolution":{"observed_at":"2026-08-12T21:27:16.675226Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:18.562150Z","title":"Egocentric video-language pretraining","venue":null,"work_id":"1225a86b-0f2f-4f4a-bb4f-7ba9cc8094bb","year":2022},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.680256Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:d2bf75887d0a45a6dac6e1be1606af237ad2cab03d45eb675d9aad1f46991681","observation_id":"756366c4-920a-4fa0-adff-8122bda308df","resolution":{"observed_at":"2026-08-12T21:27:18.569353Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.05499","last_updated":"2024-07-19T06:00:41Z","snapshot_observed_at":"2026-07-06T15:00:58.804337Z","submitted_at":"2023-03-09T18:52:16Z","title":"Grounding DINO: Marrying DINO with Grounded Pre-Training for Open-Set Object Detection","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.05499","snapshot_observed_at":"2026-08-12T21:27:16.685434Z","title":"Grounding dino: Marrying dino with grounded pre-training for open-set object detection","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.685434Z"},"links":{"cited_paper":"/paper/2303.05499","citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:d6dfc805ffcf7470d69710741d792583bbd1c7953e19a13e028c68f30daa08c7","observation_id":"cadd1a59-85f1-4e16-aa5a-d1d422355b18","resolution":{"observed_at":"2026-08-12T21:27:16.685434Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1608.03983","last_updated":"2017-05-03T16:28:09Z","snapshot_observed_at":"2026-07-06T05:06:55.589962Z","submitted_at":"2016-08-13T13:46:05Z","title":"SGDR: Stochastic Gradient Descent with Warm Restarts","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1608.03983","snapshot_observed_at":"2026-08-12T21:27:16.691686Z","title":"Sgdr: Stochastic gradient descent with warm restarts","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.691686Z"},"links":{"cited_paper":"/paper/1608.03983","citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:6c8eb37265da0fcb103abcaa8e552b4add0d2d82679428b449a26a965c92b5ea","observation_id":"d116433c-7261-411f-861d-40dd7605c0ca","resolution":{"observed_at":"2026-08-12T21:27:16.691686Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1711.05101","last_updated":"2019-01-04T21:01:49Z","snapshot_observed_at":"2026-08-09T20:34:52.923500Z","submitted_at":"2017-11-14T14:24:06Z","title":"Decoupled Weight Decay Regularization","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1711.05101","snapshot_observed_at":"2026-08-12T21:27:16.697244Z","title":"Decoupled weight decay regularization","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.697244Z"},"links":{"cited_paper":"/paper/1711.05101","citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:e174451af26b9d28b120a7566e9078e8ebee5933f40b2e14aa6a14b59a0c7183","observation_id":"30c3a2d4-49dd-4644-9a7e-f73abf1a3b68","resolution":{"observed_at":"2026-08-12T21:27:16.697244Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:18.541297Z","title":"Story-driven summariza- tion for egocentric video","venue":null,"work_id":"0d18bba4-6de9-4f9b-8661-d4533f4cd747","year":2013},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.702606Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:dcb45a3c5bbbbcc3321fd05d168d7c0c0f495924edab8c140767b3574122a5cc","observation_id":"5fefd853-04f0-49ae-bd8c-3265e7192f38","resolution":{"observed_at":"2026-08-12T21:27:18.549057Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.18386","last_updated":"2025-04-22T13:23:34Z","snapshot_observed_at":"2026-08-12T23:13:41.357483Z","submitted_at":"2024-12-24T12:16:43Z","title":"Switch-a-View: View Selection Learned from Unlabeled In-the-wild Videos","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.18386","snapshot_observed_at":"2026-08-12T21:27:16.709927Z","title":"Switch-a-view: Few-shot view selection learned from edited videos","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.709927Z"},"links":{"cited_paper":"/paper/2412.18386","citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:185dadbc067b4ee97c31e00d4d33f5d467a25684081eaba9c14aa54e89f61bbc","observation_id":"a9ae5bc4-44f2-43e1-b13f-e77415b31e02","resolution":{"observed_at":"2026-08-12T21:27:16.709927Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:18.519942Z","title":"Video summarization via multi- view representative selection","venue":null,"work_id":"309630f7-4d4d-467f-9f89-53b1ccefc186","year":2017},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.715943Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:02a044a0e4038cab1ab1c69162ba1e7616a49ffc39d87d812701003d4dcf95c5","observation_id":"04784fba-8090-4375-bf5d-e45497f5a31e","resolution":{"observed_at":"2026-08-12T21:27:18.525516Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:18.489162Z","title":"Howto100m: Learning a text-video embedding by watching hundred million narrated video clips","venue":null,"work_id":"61e7f163-3347-4789-9dfc-024cc8961744","year":2019},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.721421Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:cfe272cf9567e283983d8a755d24189e310910d9915c2cb35d7299e20f6912ee","observation_id":"fe0807fc-36ba-4265-b1fc-4fa9b5e49f6a","resolution":{"observed_at":"2026-08-12T21:27:18.496074Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.726553Z","title":"Srinivasan, Matthew Tancik, Jonathan T","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.726553Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:f2aac42a75e6810ae3136378d77bc4ebc4503fc9d43d8164c79d5e33b2c2de68","observation_id":"8106cd4f-a3de-45d2-9b5d-29a59d8cef9b","resolution":{"observed_at":"2026-08-12T21:27:16.726553Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:18.446155Z","title":"Automatized summarization of multi- player games","venue":null,"work_id":"f0388c32-2d7d-4c9d-8d2e-bf1d36ead5ee","year":null},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.731426Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:230e972cee4bd54196993b6cec418235631336547c58c77dff8bf1350fcfdfbb","observation_id":"03ae2a1e-6682-4d9f-a1ec-1b14efc1a1e2","resolution":{"observed_at":"2026-08-12T21:27:18.454733Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:18.402279Z","title":"Egoenv: Human- centric environment representations from egocentric video","venue":null,"work_id":"8d29c179-1af8-4966-9803-c580df8936ed","year":2023},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.745576Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:5bd685afe068324e101254dfec9c0e0258d8529dc48682c85fbea01fba47976b","observation_id":"61da99fb-d31c-4453-92b7-c0c58c127885","resolution":{"observed_at":"2026-08-12T21:27:18.410732Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:18.374954Z","title":"Tl; dw? summarizing instructional videos with task relevance and cross-modal saliency","venue":null,"work_id":"1a085b25-6d11-4714-81e6-9902b72f5f7d","year":2022},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.750940Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:6b4f0e43559059e50f6e9aa347c5eff9b3e5c27a74329f01236dc4185b4b2e87","observation_id":"67c3248e-d510-4503-9440-8e7f5cc3a3bb","resolution":{"observed_at":"2026-08-12T21:27:18.385894Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:18.350793Z","title":"Adaptive skip intervals: Temporal abstraction for recurrent dynamical models","venue":null,"work_id":"c394d24c-6d22-412e-afd7-0dcadae7029b","year":2018},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.756958Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:196ef9e6b54c1b3ac5a0a35b39efc0de89662f9d8b9f2f26fead16a97964b219","observation_id":"82e5f804-aefd-4974-b8dc-ef24f38a6c01","resolution":{"observed_at":"2026-08-12T21:27:18.357873Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:18.330411Z","title":"Au- tomatic video summarization by graph modeling","venue":null,"work_id":"9ba41228-c339-4821-a544-f94367cfc06c","year":2003},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.763234Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:8a3f30e48b7bbb18337099e68ee4beda92f19aa76a163483cb950b59d89ef096","observation_id":"3e63b266-ec5b-4f6c-abe9-3c919f19deda","resolution":{"observed_at":"2026-08-12T21:27:18.336833Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:18.307919Z","title":"Collabora- tive summarization of topic-related videos","venue":null,"work_id":"a55202f2-d40a-4264-b2a8-ccf570447f90","year":2017},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.771983Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:5fa152fbb7ae78da1629c4531c1cd8584835f68a73fe823f52856efee734f4b3","observation_id":"ad9bab32-aae4-4415-8174-2a602a225541","resolution":{"observed_at":"2026-08-12T21:27:18.316696Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:18.288571Z","title":"Multi-view surveillance video summarization via joint embedding and sparse optimization","venue":null,"work_id":"d0957fdd-5c26-45e6-932a-136d0763d479","year":2010},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.778053Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:eedcadf404943467023e5c2593661d290e403137e77731c2194ec3b91bbc9158","observation_id":"1deecac3-7788-4b32-ab77-5d47a72a0932","resolution":{"observed_at":"2026-08-12T21:27:18.295299Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:18.268290Z","title":"Roy-Chowdhury","venue":null,"work_id":"fe1484b7-8320-42a1-97dc-210d1f32ed09","year":2017},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.783830Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:146d6f23366645fcffbb371c7432028d1f3054552b93ae00fdd21721b11bbb8d","observation_id":"8f1a830f-752c-4357-a063-ac2da8b6188e","resolution":{"observed_at":"2026-08-12T21:27:18.274325Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:18.252163Z","title":"Bleu: a method for automatic evaluation of machine translation","venue":null,"work_id":"fd3cf8ac-07c4-44f9-8c77-6ba51dec67f4","year":2002},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.789486Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:523ed63808848f00c67f32300a74248ad9e94993c583d98d5c6a5b915e72b873","observation_id":"c21a3630-06c2-4b52-af5b-4584cb1ac0f1","resolution":{"observed_at":"2026-08-12T21:27:18.257179Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:18.232407Z","title":"Sumgraph: Video summarization via recursive graph modeling","venue":null,"work_id":"e4c2dbf3-0cae-4153-8cf8-cef52777c5d9","year":2020},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.795037Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:acc8158faeba8efa172cd2c711940e19005adc16e9094387654016ddfe3352f9","observation_id":"5a1a688b-0e61-4405-8da4-40835fb472d8","resolution":{"observed_at":"2026-08-12T21:27:18.238943Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:18.209438Z","title":"Egovlpv2: Egocentric video-language pre-training with fusion in the backbone","venue":null,"work_id":"5daa8f2d-8057-45d5-a514-89fa67b62cd0","year":2023},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.799995Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:3893de533fbb8eec0b25ccf79a437ee9ae595fface1cd72cdf24a44ed0159d7a","observation_id":"7d7ac8df-72f4-4c62-9db1-cc869eb1d79d","resolution":{"observed_at":"2026-08-12T21:27:18.215657Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:18.187817Z","title":"Vloc- net++: Deep multitask learning for semantic visual localiza- tion and odometry","venue":null,"work_id":"8a37e01f-6958-4a14-8a34-002e7161b040","year":2018},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.805015Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:f36b99ead3c219c586800f324a07db2ef7ad70dc13777ec3bf035a1bfc517d59","observation_id":"378c0f34-b5b3-4334-b4f1-51e9efeac2e5","resolution":{"observed_at":"2026-08-12T21:27:18.193781Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:18.171261Z","title":"Sidekick policy learning for active visual exploration","venue":null,"work_id":"63bddafa-32d9-4ea0-a55d-c59b136c87a9","year":2018},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.810155Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:1dc2e90a59e608101ad75bf6d37b1a230a961b9b8bf7e587f8b0ca23466e8d7b","observation_id":"a4a08e10-1781-4d09-a398-42aed2923331","resolution":{"observed_at":"2026-08-12T21:27:18.176412Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:18.155042Z","title":"Emergence of exploratory look-around behaviors through active observation completion","venue":null,"work_id":"1d30e913-b3c0-4a66-9672-6ef4314e2b87","year":2019},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.814977Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:c850b55bbe89d46ec9ed55af7d79ee353553383442983c07c358da8dfaddc3fd","observation_id":"af605da0-9852-40f0-bac3-870705489a30","resolution":{"observed_at":"2026-08-12T21:27:18.160511Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:18.138359Z","title":"Naq: Leveraging narrations as queries to super- vise episodic memory","venue":null,"work_id":"2cf78ac6-5875-47f0-9a1f-dba3a11bf329","year":2023},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":90,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.820501Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:7a1da76d1b7e3b125da3f2c178070e0932e11b43fa34f8dac0adf5fe4c210a89","observation_id":"e496eb70-24af-4f73-94f6-c2f539eb4e2a","resolution":{"observed_at":"2026-08-12T21:27:18.143654Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:18.121068Z","title":"Video summarization by learning from unpaired data","venue":null,"work_id":"bdf1f7a8-954c-49c9-b09c-62732aba3825","year":2019},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":91,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.829184Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:ba3a3e5cf6ad66968e85ca89526d3c2a37687c3f151d0e971e0451c7c5606a79","observation_id":"d19f8cfa-e3cc-4253-ad30-7ea6ae3fa489","resolution":{"observed_at":"2026-08-12T21:27:18.126653Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.835818Z","title":"Adaptive video highlight detection by learning from user history","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.835818Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:904d86f1c079e10439d1685416266621d14200c3df98f975fc2bb50296a9f148","observation_id":"0d4725b1-c25f-457f-9870-5f756309db84","resolution":{"observed_at":"2026-08-12T21:27:16.835818Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:18.092546Z","title":"Chowdhury","venue":null,"work_id":"e13412bb-f291-429a-8875-929608a9cdf9","year":2018},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":93,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.841151Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:5c913eaf4009c9ac8bfc977115926ded952d3b9b60f2a66813d528d77f3b8bb8","observation_id":"2fc92f75-2740-4f13-a654-84a748667606","resolution":{"observed_at":"2026-08-12T21:27:18.097889Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:16.846399Z","title":"Attend and segment: Attention guided active semantic segmentation","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.846399Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:5057963a20f62b15eee7b75b6672ea89bf41ae41dc1cde24f839054a875a6032","observation_id":"c5a9690f-59df-487e-b0b1-bb2c99d7e9e3","resolution":{"observed_at":"2026-08-12T21:27:16.846399Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:18.064516Z","title":"Glimpse- attend-and-explore: Self-attention for active visual explo- ration","venue":null,"work_id":"a8cc0b43-4ecc-45d6-8ea4-987c85282cf7","year":2021},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":95,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.851964Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:2810db60ff598253eef4740d76de92d23f77c27d9780fc809a5b7ebb0137443d","observation_id":"121c8f1a-836b-4abb-a24b-3acd089351ec","resolution":{"observed_at":"2026-08-12T21:27:18.070161Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:18.046566Z","title":"Actor and observer: Joint modeling of first and third-person videos","venue":null,"work_id":"29b5912f-8738-4e05-9aee-ae1884bc3021","year":2018},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":96,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.858613Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:923942c2271706273a5d0ceb6e5819e25f07ffd819d5f8bc1542ac9486e2db41","observation_id":"d6b9d89a-40b3-4049-8dfd-16e538200d9a","resolution":{"observed_at":"2026-08-12T21:27:18.052780Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:18.029253Z","title":"Tvsum: Summarizing web videos using titles","venue":null,"work_id":"bbba6d49-f813-409c-975c-c5fbfe922965","year":2015},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":97,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.863934Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:809fb9cb8e258bf0bd128f385827c20d5b2337c999c42dbe0f5abc71222e05dd","observation_id":"8481d850-ddf3-403a-8c8c-56f49059cce9","resolution":{"observed_at":"2026-08-12T21:27:18.034387Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:18.010191Z","title":"Making 360 ° video watchable in 2d: Learning videography for click free view- ing","venue":null,"work_id":"1d195d5d-2012-40ef-be1c-21e2059689ef","year":2017},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":98,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.872398Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:cb07040994406d992c690946fc64411c5e305c02cad2af3e113c70f465c82897","observation_id":"6266f243-3dcb-4190-a5e2-0dda93bef23a","resolution":{"observed_at":"2026-08-12T21:27:18.015861Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:17.989207Z","title":"Pano2vid: Automatic cinematography for watching 360 videos","venue":null,"work_id":"ff6aef1a-f056-41c3-b682-afc62123442b","year":2016},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":99,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.878297Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:fde7e9d0fc481db8d6441211b3ae3a5c1104b3c2923fecc7f60cebeb5dddaa4e","observation_id":"67a9f5c6-e0a6-41cf-9c32-e40ba2a166ae","resolution":{"observed_at":"2026-08-12T21:27:17.994590Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:27:17.968863Z","title":"Automatic con- cept discovery from parallel text and visual corpora","venue":null,"work_id":"f21fe035-6211-4deb-85f0-605e55872aa4","year":2015},"citing_paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos","version":4},"reference_index":100,"source":"pdf_text","source_observed_at":"2026-08-12T21:27:16.884770Z"},"links":{"citing_paper":"/paper/2411.08753"},"observation_digest":"sha256:b5c5dc6b9db685652caa83c4186df8fabb44b5df188121f7652d25eb530ff064","observation_id":"fbe32558-5b46-4a57-9a06-394c4ba8a462","resolution":{"observed_at":"2026-08-12T21:27:17.975142Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2411.08753","last_updated":"2025-04-10T02:02:49Z","latest_version":4,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-14T09:17:32.120373Z","submitted_at":"2024-11-13T16:31:08Z","title":"Which Viewpoint Shows it Best? Language for Weakly Supervising View Selection in Multi-view Instructional Videos"},"reference_resolution":{"displayed":100,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":72,"verified_exact":1,"verified_fuzzy":27},"total_outbound_references":132},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"thesis":"As of 14 August 2026, this Paper Citation Record lists 100 of 132 outbound references and 2 inbound Pith citation observations for arXiv:2411.08753."}