{"as_of":"2026-08-14T18:43:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:0b77b2232c043d6177696a5c5daee94ce52b5836f594818bd7b633f485406969","coverage":[{"denominator":32,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":32,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T11:29:29.724751Z","state":"measured"},{"denominator":33,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":33,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-14T06:32:32.682623+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-26T01:50:54.242508Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T15:09:55.213452Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.02356","last_updated":"2025-08-18T07:41:54Z","snapshot_observed_at":"2026-08-09T00:23:41.166151Z","submitted_at":"2025-06-03T01:16:13Z","title":"InterRVOS: Interaction-aware Referring Video Object Segmentation","version":3},"cited_work":{"arxiv_id":"2506.02356","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.02356","snapshot_observed_at":"2026-07-04T15:09:55.213452Z","title":null,"venue":null,"work_id":"bc275eef-9f90-45ce-b9ab-4b955923ff38","year":2025},"citing_paper":{"arxiv_id":"2606.26196","last_updated":"2026-06-24T15:20:32Z","snapshot_observed_at":"2026-08-13T03:40:06.714401Z","submitted_at":"2026-06-24T15:20:32Z","title":"From Structure to Synergy: A Survey of Vision-Language Perception Paradigm Evolution in Multimodal Large Language Models","version":1},"reference_index":90,"source":"pdf_text","source_observed_at":"2026-06-26T01:50:54.242508Z"},"links":{"cited_paper":"/paper/2506.02356","citing_paper":"/paper/2606.26196"},"observation_digest":"sha256:4cd98a8e85631bb0e616e6eb1ff4deb02563a7fa536566c719770c8b59da016c","observation_id":"66f5d8ba-b7bd-4887-b54e-b3e839842af9","resolution":{"observed_at":"2026-07-04T15:09:55.214944Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2506.02356/citation-record","integrity":"/paper/2506.02356/integrity","json":"/paper/2506.02356/citation-record.json","paper":"/paper/2506.02356"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:29:34.179535Z","title":"One token to seg them all: Language instructed reasoning segmentation in videos","venue":null,"work_id":"6f02f332-cb9c-4f73-a526-c2719dc65a76","year":2024},"citing_paper":{"arxiv_id":"2506.02356","last_updated":"2025-08-18T07:41:54Z","snapshot_observed_at":"2026-08-09T00:23:41.166151Z","submitted_at":"2025-06-03T01:16:13Z","title":"InterRVOS: Interaction-aware Referring Video Object Segmentation","version":3},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-07T11:29:27.623640Z"},"links":{"citing_paper":"/paper/2506.02356"},"observation_digest":"sha256:d95a440391492b30def3f233adb1539ddf2e6aeefc2ec944cb7d1231c5c9e2de","observation_id":"dee763a0-162e-4fdf-803b-272cdbdf7042","resolution":{"observed_at":"2026-08-07T11:29:34.253558Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:29:33.958931Z","title":"End-to-end referring video object segmentation with multimodal transformers","venue":null,"work_id":"931ea38f-7073-4091-a04f-b82c4a2f058c","year":2022},"citing_paper":{"arxiv_id":"2506.02356","last_updated":"2025-08-18T07:41:54Z","snapshot_observed_at":"2026-08-09T00:23:41.166151Z","submitted_at":"2025-06-03T01:16:13Z","title":"InterRVOS: Interaction-aware Referring Video Object Segmentation","version":3},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-07T11:29:27.674228Z"},"links":{"citing_paper":"/paper/2506.02356"},"observation_digest":"sha256:88953818788f74c6eb18e520a8aeee67eebb4cc4d464fc5cd2c739cf25102c91","observation_id":"85184742-8ac3-4641-a574-aecbcd50b8c5","resolution":{"observed_at":"2026-08-07T11:29:34.057605Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.05271","last_updated":"2025-09-26T12:52:41Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-06T18:57:08Z","title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.05271","snapshot_observed_at":"2026-08-07T11:29:27.777419Z","title":"Expanding performance boundaries of open-source multimodal models with model, data, and test-time scaling","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02356","last_updated":"2025-08-18T07:41:54Z","snapshot_observed_at":"2026-08-09T00:23:41.166151Z","submitted_at":"2025-06-03T01:16:13Z","title":"InterRVOS: Interaction-aware Referring Video Object Segmentation","version":3},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-07T11:29:27.777419Z"},"links":{"cited_paper":"/paper/2412.05271","citing_paper":"/paper/2506.02356"},"observation_digest":"sha256:2fcf8511481514a420f508a787432dccc012d474dcf8ee1130bab1b5673c0d37","observation_id":"5a381820-99cb-45bc-9192-783b00a298f8","resolution":{"observed_at":"2026-08-07T11:29:27.777419Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:29:33.776737Z","title":"Vision-language transformer and query generation for referring segmentation","venue":null,"work_id":"bc2b0c07-907f-4327-a778-3472acfbf3d6","year":2021},"citing_paper":{"arxiv_id":"2506.02356","last_updated":"2025-08-18T07:41:54Z","snapshot_observed_at":"2026-08-09T00:23:41.166151Z","submitted_at":"2025-06-03T01:16:13Z","title":"InterRVOS: Interaction-aware Referring Video Object Segmentation","version":3},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-07T11:29:27.834762Z"},"links":{"citing_paper":"/paper/2506.02356"},"observation_digest":"sha256:c380a3fcf3413372a9c0b733cb0b91895f30e832e760d4deaf9247a2e6969391","observation_id":"9368ec74-88b7-4f2f-a2c8-0c690f01fbda","resolution":{"observed_at":"2026-08-07T11:29:33.862292Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:29:33.564577Z","title":"Mevis: A large-scale benchmark for video segmentation with motion expressions","venue":null,"work_id":"2cb716b5-c0ca-4b5f-a4f8-e8a041393e4f","year":2023},"citing_paper":{"arxiv_id":"2506.02356","last_updated":"2025-08-18T07:41:54Z","snapshot_observed_at":"2026-08-09T00:23:41.166151Z","submitted_at":"2025-06-03T01:16:13Z","title":"InterRVOS: Interaction-aware Referring Video Object Segmentation","version":3},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-07T11:29:27.917308Z"},"links":{"citing_paper":"/paper/2506.02356"},"observation_digest":"sha256:7b2351122657967a70f53e7b1b32a38aed71ad0d574da27a3bc95ccbaa95fa14","observation_id":"baae64bd-df07-4a70-b212-5abaec036c7b","resolution":{"observed_at":"2026-08-07T11:29:33.640172Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:29:33.368559Z","title":"Moma: A multi-object multi-action dataset for understanding human activities","venue":null,"work_id":"062e3551-995c-482c-a035-5fd074a51ae1","year":2021},"citing_paper":{"arxiv_id":"2506.02356","last_updated":"2025-08-18T07:41:54Z","snapshot_observed_at":"2026-08-09T00:23:41.166151Z","submitted_at":"2025-06-03T01:16:13Z","title":"InterRVOS: Interaction-aware Referring Video Object Segmentation","version":3},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-07T11:29:27.989653Z"},"links":{"citing_paper":"/paper/2506.02356"},"observation_digest":"sha256:7d60b2c912853b44679b79b8452368c306617c6cd65856364ba4220f31697a86","observation_id":"b91dbc1a-cbb9-4d1f-aa8b-a8e0ac7e1dd8","resolution":{"observed_at":"2026-08-07T11:29:33.410861Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:29:33.250907Z","title":"Actor and action video segmentation from a sentence","venue":null,"work_id":"1ffab4f6-ba9c-4c3c-bce1-5e8c3af08547","year":2018},"citing_paper":{"arxiv_id":"2506.02356","last_updated":"2025-08-18T07:41:54Z","snapshot_observed_at":"2026-08-09T00:23:41.166151Z","submitted_at":"2025-06-03T01:16:13Z","title":"InterRVOS: Interaction-aware Referring Video Object Segmentation","version":3},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-07T11:29:28.076797Z"},"links":{"citing_paper":"/paper/2506.02356"},"observation_digest":"sha256:af90a825c4273998b7ae26fa6eef0b8e3bf9b958deee207a4e729a72ee1eb2ec","observation_id":"bf4a6e33-31c5-4bfb-ad76-b6f4018096bb","resolution":{"observed_at":"2026-08-07T11:29:33.306269Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:29:33.040716Z","title":"The llama 3 herd of models","venue":null,"work_id":"b5ce2729-c9a9-4dd7-8872-e6f067f697f6","year":2024},"citing_paper":{"arxiv_id":"2506.02356","last_updated":"2025-08-18T07:41:54Z","snapshot_observed_at":"2026-08-09T00:23:41.166151Z","submitted_at":"2025-06-03T01:16:13Z","title":"InterRVOS: Interaction-aware Referring Video Object Segmentation","version":3},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-07T11:29:28.133567Z"},"links":{"citing_paper":"/paper/2506.02356"},"observation_digest":"sha256:d66c3f6cc3b68bda8f9abb7809da7aaf5023fa60d5c5ed648d37fcd05b22a838","observation_id":"52848177-7ddd-4880-9b09-00a850dcaec7","resolution":{"observed_at":"2026-08-07T11:29:33.101988Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:29:28.190322Z","title":"Lora: Low-rank adaptation of large language models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.02356","last_updated":"2025-08-18T07:41:54Z","snapshot_observed_at":"2026-08-09T00:23:41.166151Z","submitted_at":"2025-06-03T01:16:13Z","title":"InterRVOS: Interaction-aware Referring Video Object Segmentation","version":3},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-07T11:29:28.190322Z"},"links":{"citing_paper":"/paper/2506.02356"},"observation_digest":"sha256:68457185e79a474fa8ab452259e150b01183c90fe490513d0edefecb5f7f8c7b","observation_id":"f51f17a5-0d00-409d-b20a-94e1e8dfbb97","resolution":{"observed_at":"2026-08-07T11:29:28.190322Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.21276","last_updated":"2024-10-25T17:43:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-25T17:43:01Z","title":"GPT-4o System Card","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.21276","snapshot_observed_at":"2026-08-07T11:29:28.254083Z","title":"Gpt-4o system card","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02356","last_updated":"2025-08-18T07:41:54Z","snapshot_observed_at":"2026-08-09T00:23:41.166151Z","submitted_at":"2025-06-03T01:16:13Z","title":"InterRVOS: Interaction-aware Referring Video Object Segmentation","version":3},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-07T11:29:28.254083Z"},"links":{"cited_paper":"/paper/2410.21276","citing_paper":"/paper/2506.02356"},"observation_digest":"sha256:e58d58cfe5ed13f7444a6b133f5fbc646a495724cc02ac4ba47a04943823cfc4","observation_id":"82a39cc5-de55-4213-afa2-c3f0483ebaad","resolution":{"observed_at":"2026-08-07T11:29:28.254083Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:29:32.799594Z","title":"Action genome: Actions as compositions of spatiotemporal scene graphs","venue":null,"work_id":"1add1d07-9b8d-4ae4-a923-458c0ad4ac1d","year":2020},"citing_paper":{"arxiv_id":"2506.02356","last_updated":"2025-08-18T07:41:54Z","snapshot_observed_at":"2026-08-09T00:23:41.166151Z","submitted_at":"2025-06-03T01:16:13Z","title":"InterRVOS: Interaction-aware Referring Video Object Segmentation","version":3},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-07T11:29:28.343745Z"},"links":{"citing_paper":"/paper/2506.02356"},"observation_digest":"sha256:9592fe10d7bd09d2353c04bb129672498213d5989ec3037178287fe541afb544","observation_id":"5ad1d67c-d91c-4f6d-a940-9f0cc036ade9","resolution":{"observed_at":"2026-08-07T11:29:32.908278Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:29:32.526475Z","title":"Video object segmentation with language referring expressions","venue":null,"work_id":"9f57ba29-765f-4802-b228-3e4b86f47cdc","year":2018},"citing_paper":{"arxiv_id":"2506.02356","last_updated":"2025-08-18T07:41:54Z","snapshot_observed_at":"2026-08-09T00:23:41.166151Z","submitted_at":"2025-06-03T01:16:13Z","title":"InterRVOS: Interaction-aware Referring Video Object Segmentation","version":3},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-07T11:29:28.412310Z"},"links":{"citing_paper":"/paper/2506.02356"},"observation_digest":"sha256:bdf70ab045a5a13865e81e103a61bd828f17b1951c8719763d82087e370c4342","observation_id":"86063e0b-4cee-4535-af7a-e12a2afdbae7","resolution":{"observed_at":"2026-08-07T11:29:32.619250Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.04923","last_updated":"2025-03-25T10:08:13Z","snapshot_observed_at":"2026-08-12T22:05:27.660776Z","submitted_at":"2024-11-07T17:59:27Z","title":"VideoGLaMM: A Large Multimodal Model for Pixel-Level Visual Grounding in Videos","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.04923","snapshot_observed_at":"2026-08-07T11:29:28.478864Z","title":"Videoglamm: Video grounded language-image pretraining with masked modeling","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02356","last_updated":"2025-08-18T07:41:54Z","snapshot_observed_at":"2026-08-09T00:23:41.166151Z","submitted_at":"2025-06-03T01:16:13Z","title":"InterRVOS: Interaction-aware Referring Video Object Segmentation","version":3},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-07T11:29:28.478864Z"},"links":{"cited_paper":"/paper/2411.04923","citing_paper":"/paper/2506.02356"},"observation_digest":"sha256:bcb8780a54d02b6fe5ed01514fc2f874eb81cd6ce42b3be6d566f722a6cad500","observation_id":"3c42c50f-28d6-4f2a-a0c1-5d2179bed05c","resolution":{"observed_at":"2026-08-07T11:29:28.478864Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:29:28.562199Z","title":"Visual instruction tuning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02356","last_updated":"2025-08-18T07:41:54Z","snapshot_observed_at":"2026-08-09T00:23:41.166151Z","submitted_at":"2025-06-03T01:16:13Z","title":"InterRVOS: Interaction-aware Referring Video Object Segmentation","version":3},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-07T11:29:28.562199Z"},"links":{"citing_paper":"/paper/2506.02356"},"observation_digest":"sha256:1906386fe2baf383b80ea8958d9e606c29c9656eba1aa4ea98e9b819f87d7601","observation_id":"9dd12239-b4e5-43e3-95f1-eab0bf2524d0","resolution":{"observed_at":"2026-08-07T11:29:28.562199Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:29:32.290188Z","title":"Spectrum-guided multi-granularity referring video object segmentation","venue":null,"work_id":"c9f1d01a-17aa-4e4a-8b02-0f7d7bc2f745","year":2023},"citing_paper":{"arxiv_id":"2506.02356","last_updated":"2025-08-18T07:41:54Z","snapshot_observed_at":"2026-08-09T00:23:41.166151Z","submitted_at":"2025-06-03T01:16:13Z","title":"InterRVOS: Interaction-aware Referring Video Object Segmentation","version":3},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-07T11:29:28.655247Z"},"links":{"citing_paper":"/paper/2506.02356"},"observation_digest":"sha256:fb196aebc5244778e25b30b42a57ac8d8cec519133b5a29e72d94d7afdcff00d","observation_id":"b8e90b91-2bcd-4c5a-b288-d8694652df00","resolution":{"observed_at":"2026-08-07T11:29:32.416144Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:29:32.054619Z","title":"Refer-youtube-vos: A dataset for video object segmentation with language referring expressions","venue":null,"work_id":"11c363bc-08a6-4cb1-b55a-1c9717751acc","year":2020},"citing_paper":{"arxiv_id":"2506.02356","last_updated":"2025-08-18T07:41:54Z","snapshot_observed_at":"2026-08-09T00:23:41.166151Z","submitted_at":"2025-06-03T01:16:13Z","title":"InterRVOS: Interaction-aware Referring Video Object Segmentation","version":3},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-07T11:29:28.718185Z"},"links":{"citing_paper":"/paper/2506.02356"},"observation_digest":"sha256:3ed5b08b44dafc3d51820e8046be465f8356242884e8bbca51b5c586cd6f1ffc","observation_id":"945be114-8625-4e57-9bbe-9fc57529e925","resolution":{"observed_at":"2026-08-07T11:29:32.178119Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00714","last_updated":"2024-10-28T16:37:57Z","snapshot_observed_at":"2026-07-06T18:55:41.459417Z","submitted_at":"2024-08-01T17:00:08Z","title":"SAM 2: Segment Anything in Images and Videos","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.00714","snapshot_observed_at":"2026-08-07T11:29:28.763180Z","title":"Sam 2: Segment anything in images and videos","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02356","last_updated":"2025-08-18T07:41:54Z","snapshot_observed_at":"2026-08-09T00:23:41.166151Z","submitted_at":"2025-06-03T01:16:13Z","title":"InterRVOS: Interaction-aware Referring Video Object Segmentation","version":3},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-07T11:29:28.763180Z"},"links":{"cited_paper":"/paper/2408.00714","citing_paper":"/paper/2506.02356"},"observation_digest":"sha256:76f35f0391ff6d1e8f34e2bebfc8c252b595f2a4c2acc82ffa3bad75616f31c4","observation_id":"80d7bdab-9f61-4f66-a496-b15b3a3355e5","resolution":{"observed_at":"2026-08-07T11:29:28.763180Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:29:31.702267Z","title":"Urvos: Unified referring video object segmentation network with a large-scale benchmark","venue":null,"work_id":"426ede42-8ce7-4de1-9bbe-4facc4135a59","year":2020},"citing_paper":{"arxiv_id":"2506.02356","last_updated":"2025-08-18T07:41:54Z","snapshot_observed_at":"2026-08-09T00:23:41.166151Z","submitted_at":"2025-06-03T01:16:13Z","title":"InterRVOS: Interaction-aware Referring Video Object Segmentation","version":3},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-07T11:29:28.841080Z"},"links":{"citing_paper":"/paper/2506.02356"},"observation_digest":"sha256:1fb61f443b4a9719de4e6e2a8ea78340ca16cdf1f683f71a553ddee84c79ffd5","observation_id":"db77b5ba-1be2-4a7a-9791-269ea4c563db","resolution":{"observed_at":"2026-08-07T11:29:31.861439Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:29:31.333232Z","title":"Video relationship detection","venue":null,"work_id":"6061614a-aa57-45e3-a181-a22d5d196f76","year":2017},"citing_paper":{"arxiv_id":"2506.02356","last_updated":"2025-08-18T07:41:54Z","snapshot_observed_at":"2026-08-09T00:23:41.166151Z","submitted_at":"2025-06-03T01:16:13Z","title":"InterRVOS: Interaction-aware Referring Video Object Segmentation","version":3},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-07T11:29:28.914632Z"},"links":{"citing_paper":"/paper/2506.02356"},"observation_digest":"sha256:f19a304dfd6c98894031e7e20e7707ff1a22752e4fc4230960449b51dc13639a","observation_id":"2aa89ca2-753c-40bd-b9e7-699f0531a945","resolution":{"observed_at":"2026-08-07T11:29:31.474955Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:29:30.966195Z","title":"Annotating objects and relations in user-generated videos","venue":null,"work_id":"843b32aa-aefe-4899-9590-3e9953038620","year":2019},"citing_paper":{"arxiv_id":"2506.02356","last_updated":"2025-08-18T07:41:54Z","snapshot_observed_at":"2026-08-09T00:23:41.166151Z","submitted_at":"2025-06-03T01:16:13Z","title":"InterRVOS: Interaction-aware Referring Video Object Segmentation","version":3},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-07T11:29:28.981295Z"},"links":{"citing_paper":"/paper/2506.02356"},"observation_digest":"sha256:e1d9c24e5fcf0eb38a5d5733f597f9cef2c3ee063d4348c702901ed341995903","observation_id":"71050332-79e4-42da-8df2-789af37553ba","resolution":{"observed_at":"2026-08-07T11:29:31.094806Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:29:30.745459Z","title":"Yfcc100m: The new data in multimedia research","venue":null,"work_id":"1b1e2bbb-359f-48b6-8cf5-4ba0f551684a","year":2016},"citing_paper":{"arxiv_id":"2506.02356","last_updated":"2025-08-18T07:41:54Z","snapshot_observed_at":"2026-08-09T00:23:41.166151Z","submitted_at":"2025-06-03T01:16:13Z","title":"InterRVOS: Interaction-aware Referring Video Object Segmentation","version":3},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-07T11:29:29.039012Z"},"links":{"citing_paper":"/paper/2506.02356"},"observation_digest":"sha256:e2830c360933b874a9efbb29a3df848e404f70ee5374ccd5127b489ccf0c475c","observation_id":"e30af10b-5310-4e50-a02a-6535d18e2ab7","resolution":{"observed_at":"2026-08-07T11:29:30.865552Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.17011","last_updated":"2023-05-26T15:13:44Z","snapshot_observed_at":"2026-08-13T11:32:20.841491Z","submitted_at":"2023-05-26T15:13:44Z","title":"SOC: Semantic-Assisted Object Cluster for Referring Video Object Segmentation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.17011","snapshot_observed_at":"2026-08-07T11:29:29.102921Z","title":"Soc: Segmenting objects by categories for referring video object segmentation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02356","last_updated":"2025-08-18T07:41:54Z","snapshot_observed_at":"2026-08-09T00:23:41.166151Z","submitted_at":"2025-06-03T01:16:13Z","title":"InterRVOS: Interaction-aware Referring Video Object Segmentation","version":3},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-07T11:29:29.102921Z"},"links":{"cited_paper":"/paper/2305.17011","citing_paper":"/paper/2506.02356"},"observation_digest":"sha256:5db4c951943326c999d16a3deb486ee53d825dd294d3d619009936d207c6e700","observation_id":"5cb9c047-76ec-408a-8c66-0f6c3378604a","resolution":{"observed_at":"2026-08-07T11:29:29.102921Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.14500","last_updated":"2025-03-16T14:39:54Z","snapshot_observed_at":"2026-08-12T23:19:12.912064Z","submitted_at":"2024-07-18T17:59:17Z","title":"ViLLa: Video Reasoning Segmentation with Large Language Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.14500","snapshot_observed_at":"2026-08-07T11:29:29.149297Z","title":"Villa: Unifying vision-language segmentation tasks with large multi-modal models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02356","last_updated":"2025-08-18T07:41:54Z","snapshot_observed_at":"2026-08-09T00:23:41.166151Z","submitted_at":"2025-06-03T01:16:13Z","title":"InterRVOS: Interaction-aware Referring Video Object Segmentation","version":3},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-07T11:29:29.149297Z"},"links":{"cited_paper":"/paper/2407.14500","citing_paper":"/paper/2506.02356"},"observation_digest":"sha256:28268acda8921fc43094e0b031d8c2cc2305ec82d0315b84a136bdc0d4ac9155","observation_id":"51c3e5c4-7511-4120-a3b2-65b765c8c9b2","resolution":{"observed_at":"2026-08-07T11:29:29.149297Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:29:30.554264Z","title":"Language as queries for referring video object segmentation","venue":null,"work_id":"1f8308fa-e5c8-45d6-9dfa-5dfb8314f74b","year":2022},"citing_paper":{"arxiv_id":"2506.02356","last_updated":"2025-08-18T07:41:54Z","snapshot_observed_at":"2026-08-09T00:23:41.166151Z","submitted_at":"2025-06-03T01:16:13Z","title":"InterRVOS: Interaction-aware Referring Video Object Segmentation","version":3},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-07T11:29:29.205824Z"},"links":{"citing_paper":"/paper/2506.02356"},"observation_digest":"sha256:2cbf7f76926ce4686e16cc1a3e88c23b7a88d4f8340992cae3e63cccbc4677b7","observation_id":"e701675b-9556-4572-b972-89f1ac135800","resolution":{"observed_at":"2026-08-07T11:29:30.644265Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.09711","last_updated":"2024-05-15T21:53:54Z","snapshot_observed_at":"2026-08-13T05:15:32.116612Z","submitted_at":"2024-05-15T21:53:54Z","title":"STAR: A Benchmark for Situated Reasoning in Real-World Videos","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.09711","snapshot_observed_at":"2026-08-07T11:29:29.288871Z","title":"Star: Structured action understanding in instructional videos","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02356","last_updated":"2025-08-18T07:41:54Z","snapshot_observed_at":"2026-08-09T00:23:41.166151Z","submitted_at":"2025-06-03T01:16:13Z","title":"InterRVOS: Interaction-aware Referring Video Object Segmentation","version":3},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-07T11:29:29.288871Z"},"links":{"cited_paper":"/paper/2405.09711","citing_paper":"/paper/2506.02356"},"observation_digest":"sha256:324a8e0bdf3226dad06523795d9913945ab457b72cb10cd15855cc54d45d9339","observation_id":"759cfd64-7b93-4d6e-9e2e-1b9fa7556250","resolution":{"observed_at":"2026-08-07T11:29:29.288871Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:29:30.423132Z","title":"Visa: Reasoning video object segmentation via large language models","venue":null,"work_id":"d686bd5a-0dfb-499a-b66a-98622529288d","year":2024},"citing_paper":{"arxiv_id":"2506.02356","last_updated":"2025-08-18T07:41:54Z","snapshot_observed_at":"2026-08-09T00:23:41.166151Z","submitted_at":"2025-06-03T01:16:13Z","title":"InterRVOS: Interaction-aware Referring Video Object Segmentation","version":3},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-07T11:29:29.323770Z"},"links":{"citing_paper":"/paper/2506.02356"},"observation_digest":"sha256:84271a2f0cb7e0f1c1e0e07534ce6fdba1c7bbb4cf9564eea2a88bc182f683f7","observation_id":"3efc5d46-5ba5-4ef7-8314-11c0529b260c","resolution":{"observed_at":"2026-08-07T11:29:30.477138Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.04001","last_updated":"2025-11-03T17:35:29Z","snapshot_observed_at":"2026-08-08T01:58:42.644918Z","submitted_at":"2025-01-07T18:58:54Z","title":"Sa2VA: Marrying SAM2 with LLaVA for Dense Grounded Understanding of Images and Videos","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.04001","snapshot_observed_at":"2026-08-07T11:29:29.406573Z","title":"Sa2va: Marrying sam2 with llava for dense grounded understanding of images and videos","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.02356","last_updated":"2025-08-18T07:41:54Z","snapshot_observed_at":"2026-08-09T00:23:41.166151Z","submitted_at":"2025-06-03T01:16:13Z","title":"InterRVOS: Interaction-aware Referring Video Object Segmentation","version":3},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-07T11:29:29.406573Z"},"links":{"cited_paper":"/paper/2501.04001","citing_paper":"/paper/2506.02356"},"observation_digest":"sha256:399d293986010add3b9dd3773a75fb977d7a3d6578dfd06513af8f13ed5359f6","observation_id":"651c05f3-96ee-4b49-9b2e-fae4fe859f67","resolution":{"observed_at":"2026-08-07T11:29:29.406573Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.03645","last_updated":"2024-04-04T17:58:21Z","snapshot_observed_at":"2026-08-13T00:37:30.607375Z","submitted_at":"2024-04-04T17:58:21Z","title":"Decoupling Static and Hierarchical Motion Perception for Referring Video Segmentation","version":1},"cited_work":{"arxiv_id":"2404.03645","doi":null,"metadata_source":"pith","pith_arxiv_id":"2404.03645","snapshot_observed_at":"2026-08-07T11:29:30.106190Z","title":"Decoupling Static and Hierarchical Motion Perception for Referring Video Segmentation","venue":"cs.CV","work_id":"f8c04d34-5d62-4f11-92c1-792b75484953","year":2024},"citing_paper":{"arxiv_id":"2506.02356","last_updated":"2025-08-18T07:41:54Z","snapshot_observed_at":"2026-08-09T00:23:41.166151Z","submitted_at":"2025-06-03T01:16:13Z","title":"InterRVOS: Interaction-aware Referring Video Object Segmentation","version":3},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-07T11:29:29.474932Z"},"links":{"cited_paper":"/paper/2404.03645","citing_paper":"/paper/2506.02356"},"observation_digest":"sha256:8712247bc2148d9fde405d6164da402cfcd2375230f9465c2259caeede0de751","observation_id":"f2776946-a890-4711-adbb-6403f95f64e9","resolution":{"observed_at":"2026-08-07T11:29:30.142164Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:29:29.534945Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.02356","last_updated":"2025-08-18T07:41:54Z","snapshot_observed_at":"2026-08-09T00:23:41.166151Z","submitted_at":"2025-06-03T01:16:13Z","title":"InterRVOS: Interaction-aware Referring Video Object Segmentation","version":3},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-07T11:29:29.534945Z"},"links":{"citing_paper":"/paper/2506.02356"},"observation_digest":"sha256:753ddf49d472b9a6b2b4b979ad2af9abd4fd5bed5ad609b1fdd0e072d9dd12c3","observation_id":"9034104e-4cb4-4125-b42f-18ec53fd65aa","resolution":{"observed_at":"2026-08-07T11:29:29.534945Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:29:29.594298Z","title":"@esa (Ref","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.02356","last_updated":"2025-08-18T07:41:54Z","snapshot_observed_at":"2026-08-09T00:23:41.166151Z","submitted_at":"2025-06-03T01:16:13Z","title":"InterRVOS: Interaction-aware Referring Video Object Segmentation","version":3},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-07T11:29:29.594298Z"},"links":{"citing_paper":"/paper/2506.02356"},"observation_digest":"sha256:84fed820ede1864c8dff55613c402871b4c39034464844962350f364da22e702","observation_id":"1d30c372-044f-4464-b690-037358a9bc37","resolution":{"observed_at":"2026-08-07T11:29:29.594298Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:29:29.642842Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.02356","last_updated":"2025-08-18T07:41:54Z","snapshot_observed_at":"2026-08-09T00:23:41.166151Z","submitted_at":"2025-06-03T01:16:13Z","title":"InterRVOS: Interaction-aware Referring Video Object Segmentation","version":3},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-07T11:29:29.642842Z"},"links":{"citing_paper":"/paper/2506.02356"},"observation_digest":"sha256:4e4664dc3a8cae5cd58dd6ad671832fc968ba28afd164c28e23e245516546143","observation_id":"c13ea854-4ba4-4081-8f51-25305ab4c2b2","resolution":{"observed_at":"2026-08-07T11:29:29.642842Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:29:29.724751Z","title":"A child helping another child with a backpack","venue":null,"work_id":null,"year":2000},"citing_paper":{"arxiv_id":"2506.02356","last_updated":"2025-08-18T07:41:54Z","snapshot_observed_at":"2026-08-09T00:23:41.166151Z","submitted_at":"2025-06-03T01:16:13Z","title":"InterRVOS: Interaction-aware Referring Video Object Segmentation","version":3},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-07T11:29:29.724751Z"},"links":{"citing_paper":"/paper/2506.02356"},"observation_digest":"sha256:e8e2311b8b0b848932755f3c555e83e368a0a5473dc0479fcbf340bf0fb88fba","observation_id":"ce1d1cc0-77ce-4bad-b412-6a3faf66d064","resolution":{"observed_at":"2026-08-07T11:29:29.724751Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2506.02356","last_updated":"2025-08-18T07:41:54Z","latest_version":3,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-09T00:23:41.166151Z","submitted_at":"2025-06-03T01:16:13Z","title":"InterRVOS: Interaction-aware Referring Video Object Segmentation"},"reference_resolution":{"displayed":32,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":14,"verified_exact":1,"verified_fuzzy":17},"total_outbound_references":32},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"thesis":"As of 14 August 2026, this Paper Citation Record lists 32 of 32 outbound references and 1 inbound Pith citation observation for arXiv:2506.02356."}