{"as_of":"2026-08-10T22:52:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:1a0c207ab0ab74253b11f4285e993b1a48e0845117faf3bdab7b0547208e1162","coverage":[{"denominator":24,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":24,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T05:53:10.443418Z","state":"measured"},{"denominator":24,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":24,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2506.06748/citation-record","integrity":"/paper/2506.06748/integrity","json":"/paper/2506.06748/citation-record.json","paper":"/paper/2506.06748"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:53:10.665226Z","title":"One- shot video object segmentation","venue":null,"work_id":"1c69b9fc-2141-428b-9b61-b78f9733f462","year":null},"citing_paper":{"arxiv_id":"2506.06748","last_updated":"2025-06-07T10:33:16Z","snapshot_observed_at":"2026-08-10T21:14:08.234066Z","submitted_at":"2025-06-07T10:33:16Z","title":"THU-Warwick Submission for EPIC-KITCHEN Challenge 2025: Semi-Supervised Video Object Segmentation","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T05:53:10.369033Z"},"links":{"citing_paper":"/paper/2506.06748"},"observation_digest":"sha256:ff062b77c22a25a7e183600b70d4abfafde634c0ca706acf30ce0ecf32a4ddb4","observation_id":"19a82926-2c95-4aa3-bcea-5bd2d73fed26","resolution":{"observed_at":"2026-08-07T05:53:10.668238Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:53:10.657533Z","title":"Xmem: Long- term video object segmentation with an atkinson-shiffrin memory model","venue":null,"work_id":"62c969c8-0ea9-4542-b6ef-92b7d7eac7f0","year":2022},"citing_paper":{"arxiv_id":"2506.06748","last_updated":"2025-06-07T10:33:16Z","snapshot_observed_at":"2026-08-10T21:14:08.234066Z","submitted_at":"2025-06-07T10:33:16Z","title":"THU-Warwick Submission for EPIC-KITCHEN Challenge 2025: Semi-Supervised Video Object Segmentation","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T05:53:10.372867Z"},"links":{"citing_paper":"/paper/2506.06748"},"observation_digest":"sha256:51acdb48fb732d309080ca3fecab66762655d5203b416b8766f198c07bb921a8","observation_id":"d4b009cf-e521-4bee-b322-0214b6f4f35e","resolution":{"observed_at":"2026-08-07T05:53:10.660577Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:53:10.649914Z","title":"Rethink- ing space-time networks with improved memory coverage for efficient video object segmentation","venue":null,"work_id":"679a21df-72b4-401d-9d7e-e493950f1bb2","year":null},"citing_paper":{"arxiv_id":"2506.06748","last_updated":"2025-06-07T10:33:16Z","snapshot_observed_at":"2026-08-10T21:14:08.234066Z","submitted_at":"2025-06-07T10:33:16Z","title":"THU-Warwick Submission for EPIC-KITCHEN Challenge 2025: Semi-Supervised Video Object Segmentation","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T05:53:10.376144Z"},"links":{"citing_paper":"/paper/2506.06748"},"observation_digest":"sha256:fb9cabbd8a4947353367b27fe920d6136da93e0531982940c67569e8a94060e5","observation_id":"d25ee1b0-3f5f-43ae-91e3-888e0fe39b4a","resolution":{"observed_at":"2026-08-07T05:53:10.652929Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:53:10.641776Z","title":"Putting the object back into video object segmentation","venue":null,"work_id":"ab33dffc-6984-46d7-b388-9546974a319a","year":2024},"citing_paper":{"arxiv_id":"2506.06748","last_updated":"2025-06-07T10:33:16Z","snapshot_observed_at":"2026-08-10T21:14:08.234066Z","submitted_at":"2025-06-07T10:33:16Z","title":"THU-Warwick Submission for EPIC-KITCHEN Challenge 2025: Semi-Supervised Video Object Segmentation","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T05:53:10.379426Z"},"links":{"citing_paper":"/paper/2506.06748"},"observation_digest":"sha256:c8ef415739101b2f743515dcdef30f534fe1fae3ffa1ac8349e10d49cff78712","observation_id":"a1c78d10-fc14-446f-8dac-9c7e340d831f","resolution":{"observed_at":"2026-08-07T05:53:10.644530Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:53:10.632117Z","title":"Epic-kitchens visor benchmark: Video segmenta- tions and object relations","venue":null,"work_id":"c0793f17-2a58-40f4-adc5-3c300520cbfe","year":2022},"citing_paper":{"arxiv_id":"2506.06748","last_updated":"2025-06-07T10:33:16Z","snapshot_observed_at":"2026-08-10T21:14:08.234066Z","submitted_at":"2025-06-07T10:33:16Z","title":"THU-Warwick Submission for EPIC-KITCHEN Challenge 2025: Semi-Supervised Video Object Segmentation","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T05:53:10.382571Z"},"links":{"citing_paper":"/paper/2506.06748"},"observation_digest":"sha256:20b6b54ef7ae8e2a064c6e7c4659506548d3dacc49a3fc4434f158b352a381aa","observation_id":"941f7b66-6ae6-4f07-8b74-270d4f96a4b5","resolution":{"observed_at":"2026-08-07T05:53:10.635716Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:53:10.622568Z","title":"MOSE: A new dataset for video object segmentation in complex scenes","venue":null,"work_id":"a4e205e2-0048-4035-9bc1-187fb8d29fa8","year":null},"citing_paper":{"arxiv_id":"2506.06748","last_updated":"2025-06-07T10:33:16Z","snapshot_observed_at":"2026-08-10T21:14:08.234066Z","submitted_at":"2025-06-07T10:33:16Z","title":"THU-Warwick Submission for EPIC-KITCHEN Challenge 2025: Semi-Supervised Video Object Segmentation","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T05:53:10.385897Z"},"links":{"citing_paper":"/paper/2506.06748"},"observation_digest":"sha256:cf7f8c9dee63559ffc6f589eec66d3da2a7b24939062449845a669b68491fa66","observation_id":"a1cf217f-134b-4a30-a321-5aeec0587cf2","resolution":{"observed_at":"2026-08-07T05:53:10.625621Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.16268","last_updated":"2025-07-29T00:08:01Z","snapshot_observed_at":"2026-07-06T19:37:16.203681Z","submitted_at":"2024-10-21T17:59:19Z","title":"SAM2Long: Enhancing SAM 2 for Long Video Segmentation with a Training-Free Memory Tree","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.16268","snapshot_observed_at":"2026-08-07T05:53:10.389194Z","title":"Sam2long: Enhancing sam 2 for long video seg- mentation with a training-free memory tree.arXiv preprint arXiv:2410.16268, 2024","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.06748","last_updated":"2025-06-07T10:33:16Z","snapshot_observed_at":"2026-08-10T21:14:08.234066Z","submitted_at":"2025-06-07T10:33:16Z","title":"THU-Warwick Submission for EPIC-KITCHEN Challenge 2025: Semi-Supervised Video Object Segmentation","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T05:53:10.389194Z"},"links":{"cited_paper":"/paper/2410.16268","citing_paper":"/paper/2506.06748"},"observation_digest":"sha256:7eadca95241055398d7b54455aafc8658ec9c20cce50e254300515af0a50b23f","observation_id":"fc2dfb4c-ce12-47bb-9e93-18b824d08f2d","resolution":{"observed_at":"2026-08-07T05:53:10.389194Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:53:10.614040Z","title":"Deep learning for video object segmentation: a review.Artificial Intelligence Review, 56(1):457–531, 2023","venue":null,"work_id":"733d7633-867c-4285-83f2-3c6bd58606a9","year":2023},"citing_paper":{"arxiv_id":"2506.06748","last_updated":"2025-06-07T10:33:16Z","snapshot_observed_at":"2026-08-10T21:14:08.234066Z","submitted_at":"2025-06-07T10:33:16Z","title":"THU-Warwick Submission for EPIC-KITCHEN Challenge 2025: Semi-Supervised Video Object Segmentation","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T05:53:10.392628Z"},"links":{"citing_paper":"/paper/2506.06748"},"observation_digest":"sha256:3cbd5522086a4a37165ad193ee47b0b0b1f2b851f96635c73f8e3c31f93155ae","observation_id":"1102344f-e822-4535-961d-45abdfe6b9f2","resolution":{"observed_at":"2026-08-07T05:53:10.617011Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:53:10.603602Z","title":"Ego-exo4d: Understanding skilled human activity from first-and third-person perspectives","venue":null,"work_id":"43447279-23b9-4d0d-9ad5-a0f9c40503e5","year":2024},"citing_paper":{"arxiv_id":"2506.06748","last_updated":"2025-06-07T10:33:16Z","snapshot_observed_at":"2026-08-10T21:14:08.234066Z","submitted_at":"2025-06-07T10:33:16Z","title":"THU-Warwick Submission for EPIC-KITCHEN Challenge 2025: Semi-Supervised Video Object Segmentation","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T05:53:10.395481Z"},"links":{"citing_paper":"/paper/2506.06748"},"observation_digest":"sha256:b8c8563f9b0f1639a5a61b020c71bf2a647ca5151248177d318c48468ecc0645","observation_id":"c2a2926a-1205-499e-a68b-6ac7e00dc630","resolution":{"observed_at":"2026-08-07T05:53:10.607594Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.19326","last_updated":"2024-05-01T01:30:58Z","snapshot_observed_at":"2026-07-06T18:07:30.890899Z","submitted_at":"2024-04-30T07:50:29Z","title":"LVOS: A Benchmark for Large-scale Long-term Video Object Segmentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.19326","snapshot_observed_at":"2026-08-07T05:53:10.398572Z","title":"Lvos: A benchmark for large- scale long-term video object segmentation.arXiv preprint arXiv:2404.19326, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.06748","last_updated":"2025-06-07T10:33:16Z","snapshot_observed_at":"2026-08-10T21:14:08.234066Z","submitted_at":"2025-06-07T10:33:16Z","title":"THU-Warwick Submission for EPIC-KITCHEN Challenge 2025: Semi-Supervised Video Object Segmentation","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T05:53:10.398572Z"},"links":{"cited_paper":"/paper/2404.19326","citing_paper":"/paper/2506.06748"},"observation_digest":"sha256:5ca8ba26910e6589e9e02dd62949a3885a2f282ad3df6494e09867572e5c4b01","observation_id":"43f6b820-f8e6-443c-8814-23572813fa5d","resolution":{"observed_at":"2026-08-07T05:53:10.398572Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:53:10.594991Z","title":"Video object segmentation with adaptive feature bank and uncertain-region refinement","venue":null,"work_id":"a1114936-59a9-4e06-856a-80cf577464ac","year":2020},"citing_paper":{"arxiv_id":"2506.06748","last_updated":"2025-06-07T10:33:16Z","snapshot_observed_at":"2026-08-10T21:14:08.234066Z","submitted_at":"2025-06-07T10:33:16Z","title":"THU-Warwick Submission for EPIC-KITCHEN Challenge 2025: Semi-Supervised Video Object Segmentation","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T05:53:10.401905Z"},"links":{"citing_paper":"/paper/2506.06748"},"observation_digest":"sha256:ba1578b50c577c516c0f731d2fe834e8c3b5a1fdb5187e122ef5c74d3a542eec","observation_id":"16d95078-7ebb-454b-a0c4-3c27e8b383aa","resolution":{"observed_at":"2026-08-07T05:53:10.598545Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:53:10.587092Z","title":"Video object segmentation using space-time memory networks","venue":null,"work_id":"ce641882-4404-4ce0-81dc-da6fe314d29b","year":2019},"citing_paper":{"arxiv_id":"2506.06748","last_updated":"2025-06-07T10:33:16Z","snapshot_observed_at":"2026-08-10T21:14:08.234066Z","submitted_at":"2025-06-07T10:33:16Z","title":"THU-Warwick Submission for EPIC-KITCHEN Challenge 2025: Semi-Supervised Video Object Segmentation","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T05:53:10.404328Z"},"links":{"citing_paper":"/paper/2506.06748"},"observation_digest":"sha256:a620185bdc5cefbd1a9c7a033a3f014053be2b31084bd5fffcee2bf81f60d1bc","observation_id":"687cdb10-da89-4b1b-b2ee-628cc908583a","resolution":{"observed_at":"2026-08-07T05:53:10.590004Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:53:10.577864Z","title":null,"venue":null,"work_id":"e97296fe-373b-48e0-88bb-c9c9a980d08b","year":2023},"citing_paper":{"arxiv_id":"2506.06748","last_updated":"2025-06-07T10:33:16Z","snapshot_observed_at":"2026-08-10T21:14:08.234066Z","submitted_at":"2025-06-07T10:33:16Z","title":"THU-Warwick Submission for EPIC-KITCHEN Challenge 2025: Semi-Supervised Video Object Segmentation","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T05:53:10.406886Z"},"links":{"citing_paper":"/paper/2506.06748"},"observation_digest":"sha256:ac200ea4c473ff2730a3d13a6b79b2e6bc118a70a2b2ab4443702fcdb7dcd015","observation_id":"ff231224-9342-405b-a099-a12e4d4485d3","resolution":{"observed_at":"2026-08-07T05:53:10.581092Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:53:10.567631Z","title":"Hd-epic: A highly-detailed egocentric video dataset","venue":null,"work_id":"410d3634-f274-4fec-941c-c82e620c23e7","year":2025},"citing_paper":{"arxiv_id":"2506.06748","last_updated":"2025-06-07T10:33:16Z","snapshot_observed_at":"2026-08-10T21:14:08.234066Z","submitted_at":"2025-06-07T10:33:16Z","title":"THU-Warwick Submission for EPIC-KITCHEN Challenge 2025: Semi-Supervised Video Object Segmentation","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T05:53:10.410176Z"},"links":{"citing_paper":"/paper/2506.06748"},"observation_digest":"sha256:e04eb9843e86c3c72338e6b07967c68ef9a20915c9971857c22f9e21bf6dc465","observation_id":"ad4e75bf-c6ed-4297-a0c7-36e540a42f4e","resolution":{"observed_at":"2026-08-07T05:53:10.570757Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:53:10.558702Z","title":"An outlook into the fu- ture of egocentric vision.IJCV, 132(11):4880–4936, 2024","venue":null,"work_id":"83cf767b-8697-46a5-8b76-a1b66b9b158e","year":2024},"citing_paper":{"arxiv_id":"2506.06748","last_updated":"2025-06-07T10:33:16Z","snapshot_observed_at":"2026-08-10T21:14:08.234066Z","submitted_at":"2025-06-07T10:33:16Z","title":"THU-Warwick Submission for EPIC-KITCHEN Challenge 2025: Semi-Supervised Video Object Segmentation","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T05:53:10.413282Z"},"links":{"citing_paper":"/paper/2506.06748"},"observation_digest":"sha256:373baabcd945cb5492b22422060c2482c07edf077fa52df61e3bad57d3bc2ffb","observation_id":"e8fba540-06d8-4cbb-bba9-fb6528151569","resolution":{"observed_at":"2026-08-07T05:53:10.561663Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1704.00675","last_updated":"2018-03-01T17:50:08Z","snapshot_observed_at":"2026-08-02T10:51:13.194643Z","submitted_at":"2017-04-03T16:44:46Z","title":"The 2017 DAVIS Challenge on Video Object Segmentation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1704.00675","snapshot_observed_at":"2026-08-07T05:53:10.415724Z","title":"The 2017 davis challenge on video object segmentation.arXiv preprint arXiv:1704.00675, 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2506.06748","last_updated":"2025-06-07T10:33:16Z","snapshot_observed_at":"2026-08-10T21:14:08.234066Z","submitted_at":"2025-06-07T10:33:16Z","title":"THU-Warwick Submission for EPIC-KITCHEN Challenge 2025: Semi-Supervised Video Object Segmentation","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T05:53:10.415724Z"},"links":{"cited_paper":"/paper/1704.00675","citing_paper":"/paper/2506.06748"},"observation_digest":"sha256:04e88903680615a451221c2eaec8c0da633f3b3b7ab13c21f79e7d18a8d151d9","observation_id":"037fd328-2d0d-4774-8016-4bd121ff6e34","resolution":{"observed_at":"2026-08-07T05:53:10.415724Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:53:10.549810Z","title":"Vi- sion transformers for dense prediction","venue":null,"work_id":"c42b93a9-86ef-435f-a284-2e03d8bdcfbc","year":2021},"citing_paper":{"arxiv_id":"2506.06748","last_updated":"2025-06-07T10:33:16Z","snapshot_observed_at":"2026-08-10T21:14:08.234066Z","submitted_at":"2025-06-07T10:33:16Z","title":"THU-Warwick Submission for EPIC-KITCHEN Challenge 2025: Semi-Supervised Video Object Segmentation","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T05:53:10.419058Z"},"links":{"citing_paper":"/paper/2506.06748"},"observation_digest":"sha256:af64d548f1ccc725a3764eceb6b9be0d1cdec23c0cc28799815f0b206df3c6a9","observation_id":"f98f7708-bbaa-42eb-8627-2d312e3d8eed","resolution":{"observed_at":"2026-08-07T05:53:10.553197Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00714","last_updated":"2024-10-28T16:37:57Z","snapshot_observed_at":"2026-07-06T18:55:41.459417Z","submitted_at":"2024-08-01T17:00:08Z","title":"SAM 2: Segment Anything in Images and Videos","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.00714","snapshot_observed_at":"2026-08-07T05:53:10.423028Z","title":"Sam 2: Segment anything in images and videos.arXiv preprint arXiv:2408.00714, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.06748","last_updated":"2025-06-07T10:33:16Z","snapshot_observed_at":"2026-08-10T21:14:08.234066Z","submitted_at":"2025-06-07T10:33:16Z","title":"THU-Warwick Submission for EPIC-KITCHEN Challenge 2025: Semi-Supervised Video Object Segmentation","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T05:53:10.423028Z"},"links":{"cited_paper":"/paper/2408.00714","citing_paper":"/paper/2506.06748"},"observation_digest":"sha256:418ac7145c13bd5cbbef69a1af2c9410c72eaeaf55e5b45e10cbe6bfd0a9b251","observation_id":"c84fca61-5c02-477e-9a22-5b40a08f1b92","resolution":{"observed_at":"2026-08-07T05:53:10.423028Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:53:10.540429Z","title":"Hi- era: A hierarchical vision transformer without the bells-and- whistles","venue":null,"work_id":"6dc97f21-9573-4285-8f01-dce26afb7ef2","year":2023},"citing_paper":{"arxiv_id":"2506.06748","last_updated":"2025-06-07T10:33:16Z","snapshot_observed_at":"2026-08-10T21:14:08.234066Z","submitted_at":"2025-06-07T10:33:16Z","title":"THU-Warwick Submission for EPIC-KITCHEN Challenge 2025: Semi-Supervised Video Object Segmentation","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T05:53:10.426737Z"},"links":{"citing_paper":"/paper/2506.06748"},"observation_digest":"sha256:5f3629d30f2069385f5b46ef89733899a14ab9c8518bff5779452e1bb828117b","observation_id":"a9d641a0-7279-4079-8b35-921d99958325","resolution":{"observed_at":"2026-08-07T05:53:10.543682Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.17576","last_updated":"2024-12-04T08:58:53Z","snapshot_observed_at":"2026-07-06T19:57:24.407368Z","submitted_at":"2024-11-26T16:41:09Z","title":"A Distractor-Aware Memory for Visual Object Tracking with SAM2","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.17576","snapshot_observed_at":"2026-08-07T05:53:10.430158Z","title":"A distractor-aware memory for visual object tracking with sam2.arXiv preprint arXiv:2411.17576, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.06748","last_updated":"2025-06-07T10:33:16Z","snapshot_observed_at":"2026-08-10T21:14:08.234066Z","submitted_at":"2025-06-07T10:33:16Z","title":"THU-Warwick Submission for EPIC-KITCHEN Challenge 2025: Semi-Supervised Video Object Segmentation","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T05:53:10.430158Z"},"links":{"cited_paper":"/paper/2411.17576","citing_paper":"/paper/2506.06748"},"observation_digest":"sha256:1ac39e0ab247fa7d6b2f56649974793eda96ac77acb0293065f28b47b1baa6b3","observation_id":"290196ea-093f-41f3-87bd-c9a12f9af0f8","resolution":{"observed_at":"2026-08-07T05:53:10.430158Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:53:10.527631Z","title":"Feelvos: Fast end-to-end embedding learning for video object segmenta- tion","venue":null,"work_id":"f6b0ab9d-f745-49e9-bacd-4a6fb6a9a961","year":2019},"citing_paper":{"arxiv_id":"2506.06748","last_updated":"2025-06-07T10:33:16Z","snapshot_observed_at":"2026-08-10T21:14:08.234066Z","submitted_at":"2025-06-07T10:33:16Z","title":"THU-Warwick Submission for EPIC-KITCHEN Challenge 2025: Semi-Supervised Video Object Segmentation","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T05:53:10.434401Z"},"links":{"citing_paper":"/paper/2506.06748"},"observation_digest":"sha256:fa614f6f4ea09e0060e88e7c640e249279e6ab2920247ea6e6a62f0ad74768b2","observation_id":"70887bb4-34a3-4fe3-af64-a5ec28897730","resolution":{"observed_at":"2026-08-07T05:53:10.533170Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1809.03327","last_updated":"2018-09-06T04:19:45Z","snapshot_observed_at":"2026-07-06T07:00:12.122241Z","submitted_at":"2018-09-06T04:19:45Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1809.03327","snapshot_observed_at":"2026-08-07T05:53:10.437191Z","title":"Youtube-vos: A large-scale video object segmentation benchmark.arXiv preprint arXiv:1809.03327, 2018","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2506.06748","last_updated":"2025-06-07T10:33:16Z","snapshot_observed_at":"2026-08-10T21:14:08.234066Z","submitted_at":"2025-06-07T10:33:16Z","title":"THU-Warwick Submission for EPIC-KITCHEN Challenge 2025: Semi-Supervised Video Object Segmentation","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T05:53:10.437191Z"},"links":{"cited_paper":"/paper/1809.03327","citing_paper":"/paper/2506.06748"},"observation_digest":"sha256:ecec17948a4e5f691f91732c8586196ba772907cb464c5e495d491daef982ab5","observation_id":"68d8fad7-d0e9-4b5d-bb39-930d446d4197","resolution":{"observed_at":"2026-08-07T05:53:10.437191Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.11922","last_updated":"2024-11-30T22:32:34Z","snapshot_observed_at":"2026-08-08T21:51:38.855361Z","submitted_at":"2024-11-18T05:59:03Z","title":"SAMURAI: Adapting Segment Anything Model for Zero-Shot Visual Tracking with Motion-Aware Memory","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.11922","snapshot_observed_at":"2026-08-07T05:53:10.440098Z","title":"Samurai: Adapting segment anything model for zero-shot visual tracking with motion-aware memory.arXiv preprint arXiv:2411.11922,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.06748","last_updated":"2025-06-07T10:33:16Z","snapshot_observed_at":"2026-08-10T21:14:08.234066Z","submitted_at":"2025-06-07T10:33:16Z","title":"THU-Warwick Submission for EPIC-KITCHEN Challenge 2025: Semi-Supervised Video Object Segmentation","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T05:53:10.440098Z"},"links":{"cited_paper":"/paper/2411.11922","citing_paper":"/paper/2506.06748"},"observation_digest":"sha256:a841337d9be879c3ca75716a14566e77785bd08334bc5245518f53bba142ca27","observation_id":"1f06af49-3be1-4c6d-b98c-c8c3f86b6ab7","resolution":{"observed_at":"2026-08-07T05:53:10.440098Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.09414","last_updated":"2024-10-20T11:24:09Z","snapshot_observed_at":"2026-07-06T18:30:32.982860Z","submitted_at":"2024-06-13T17:59:56Z","title":"Depth Anything V2","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.09414","snapshot_observed_at":"2026-08-07T05:53:10.443418Z","title":"Depth any- thing v2.arXiv:2406.09414, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.06748","last_updated":"2025-06-07T10:33:16Z","snapshot_observed_at":"2026-08-10T21:14:08.234066Z","submitted_at":"2025-06-07T10:33:16Z","title":"THU-Warwick Submission for EPIC-KITCHEN Challenge 2025: Semi-Supervised Video Object Segmentation","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T05:53:10.443418Z"},"links":{"cited_paper":"/paper/2406.09414","citing_paper":"/paper/2506.06748"},"observation_digest":"sha256:00361ebcf4c3bba6ad5b716cfeb2bf80eebd63d92aa9be6f45f9045e1790b727","observation_id":"8407f388-274e-47a0-91f8-bb5cbced8cab","resolution":{"observed_at":"2026-08-07T05:53:10.443418Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2506.06748","last_updated":"2025-06-07T10:33:16Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-10T21:14:08.234066Z","submitted_at":"2025-06-07T10:33:16Z","title":"THU-Warwick Submission for EPIC-KITCHEN Challenge 2025: Semi-Supervised Video Object Segmentation"},"reference_resolution":{"displayed":24,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":9,"verified_exact":0,"verified_fuzzy":15},"total_outbound_references":24},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 24 of 24 outbound references and 0 inbound Pith citation observations for arXiv:2506.06748."}