{"as_of":"2026-08-17T04:19:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:96f69377eba60e6ee73345beacddc9d8455fe2c6312d448a9fcf3d4ad35b035f","coverage":[{"denominator":60,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":60,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T14:47:27.120881Z","state":"measured"},{"denominator":60,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":60,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-16T06:30:59.297886+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2505.17695/citation-record","integrity":"/paper/2505.17695/integrity","json":"/paper/2505.17695/citation-record.json","paper":"/paper/2505.17695"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-17T03:30:12.735656Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-07T14:47:20.053296Z","title":"Gpt-4 technical report","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:20.053296Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:a3a79eb78cd504b49e490faf5b4d01eb26bbbd2dc853e8b6db0fdeb95c8e74f2","observation_id":"18aa0765-f4fc-43f3-9784-94f3d02bdcaf","resolution":{"observed_at":"2026-08-07T14:47:20.053296Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:40.042876Z","title":"Vqa: Visual question answering","venue":null,"work_id":"7dea7b19-251b-40e6-9144-b5196c11b1a5","year":null},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:20.224226Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:4676c8d9635a51581356aae2ba8c686dd9a81860a702111e01d81ce425aae955","observation_id":"db9ff230-e126-4178-a790-947e329204ee","resolution":{"observed_at":"2026-08-07T14:47:40.174662Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:39.906864Z","title":"Coco- stuff: Thing and stuff classes in context","venue":null,"work_id":"5a608b8d-202e-4ec7-9d0f-9848b2aa0d6d","year":2018},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:20.354744Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:d1c021d34bd7e0fb26aa19b7edbc7b30b444e2bce87f56801ef6fdae579756b7","observation_id":"75d2d425-9aed-44f3-bff1-77c513afcb5e","resolution":{"observed_at":"2026-08-07T14:47:39.965237Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:39.771808Z","title":"Detect what you can: De- tecting and representing objects using holistic models and body parts","venue":null,"work_id":"9f6e34ec-1ccf-4aae-bf31-8d26462a8d9a","year":1971},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:20.582089Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:63f453f6b86e873eb0a86ab7c2f1b86c02c539aa15baeed7b1fddf86944068e6","observation_id":"0040dbe8-b07a-4196-b433-0c795bcf9153","resolution":{"observed_at":"2026-08-07T14:47:39.855083Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:39.504157Z","title":"Sam4mllm: Enhance multi- modal large language model for referring expression seg- mentation","venue":null,"work_id":"cfb634e8-de9b-493f-a7bf-a1394f5473a1","year":2024},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:20.719249Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:a7cc2a3aad24258a2ea4b104dbf37febbb3868855c4665609d4ab0e99bb6c8fb","observation_id":"29e63728-7e7c-4d1d-840a-d251c80e6c04","resolution":{"observed_at":"2026-08-07T14:47:39.662221Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:39.284318Z","title":"The cityscapes dataset for semantic urban scene understanding","venue":null,"work_id":"0b0c2ae4-9e1a-4dfe-a2f9-6a2c2683a8d9","year":2016},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:20.814327Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:6cdf9a025a101011daec495058ecff73d19e0b04f125a6702344747c02fe80fc","observation_id":"593d3b04-b604-4091-b064-ecf33998db1b","resolution":{"observed_at":"2026-08-07T14:47:39.388987Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:38.992390Z","title":"Divergen: Improving instance segmentation by learning wider data distribution with more diverse generative data","venue":null,"work_id":"44bf5f59-6b9d-44bf-9f90-bf9abcba0c3f","year":null},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:20.945150Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:af0469a13de4f14a73918b5442f357ad76259843de93aec31a94dc9b702bc33e","observation_id":"ad7a91b9-c475-43c5-a60d-ee70e36b05ab","resolution":{"observed_at":"2026-08-07T14:47:39.171952Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:38.764887Z","title":"Finding nemo: Negative- mined mosaic augmentation for referring image segmentation","venue":null,"work_id":"855726e2-aef2-47c2-8127-f8e6f49e8822","year":2024},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:21.055373Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:a77cbbd2da84adf0823703df5beaa95e39d764ad4efbc68b86bfda795af53387","observation_id":"4f51af55-ece7-4008-92f8-d16910be9951","resolution":{"observed_at":"2026-08-07T14:47:38.874547Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:38.459396Z","title":"Mixgen: A new multi- modal data augmentation","venue":null,"work_id":"f7db3447-aab3-4b28-afc5-71b6822adf1a","year":2023},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:21.156345Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:b6fdaf6556e26ff178ed7e7e85372ddea756de206b83014357bf419de0e888ea","observation_id":"ed78cdbe-d00e-4fc0-ba65-b58e6c349938","resolution":{"observed_at":"2026-08-07T14:47:38.621618Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:38.220948Z","title":"Partimagenet: A large, high-quality dataset of parts","venue":null,"work_id":"4f3e6a6c-7696-458f-8571-5b87f570a90d","year":2022},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:21.270993Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:d1d886a976f4a972cbf1536abbcf3d89e8415fad8b1a4ca3f870e77b5da9263a","observation_id":"6b5d6fab-f880-40aa-ab36-6cfdcd249263","resolution":{"observed_at":"2026-08-07T14:47:38.344668Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:21.417306Z","title":"Denoising dif- fusion probabilistic models","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:21.417306Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:fe364fa4cfb6d195bfb92a74b91ed7eda48efd2a9a2d3093f959595708e7921e","observation_id":"4d663812-4d0e-492c-a258-9c9a1b0d5a54","resolution":{"observed_at":"2026-08-07T14:47:21.417306Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:37.955024Z","title":"Beyond one-to-one: Rethinking the referring image segmentation","venue":null,"work_id":"ea54812d-f07f-42ff-9d89-473d717230b4","year":null},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:21.520435Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:f20d9139030237d970c2f2249a4042d32a4d0720f6fd27d6656e5f0377d23305","observation_id":"cc5a2dd0-2c42-4778-800d-c7a349015700","resolution":{"observed_at":"2026-08-07T14:47:38.073131Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.21276","last_updated":"2024-10-25T17:43:01Z","snapshot_observed_at":"2026-08-15T14:02:47.366139Z","submitted_at":"2024-10-25T17:43:01Z","title":"GPT-4o System Card","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.21276","snapshot_observed_at":"2026-08-07T14:47:21.632881Z","title":"Gpt-4o system card","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:21.632881Z"},"links":{"cited_paper":"/paper/2410.21276","citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:838f5181ed33ed93f53b29c1afa2c1fab22dfda9eab714fd540fe18cf975b53c","observation_id":"982feeb2-8e80-48da-ad7e-2911838a7c6b","resolution":{"observed_at":"2026-08-07T14:47:21.632881Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.10086","last_updated":"2024-08-19T15:27:25Z","snapshot_observed_at":"2026-08-16T13:25:29.150906Z","submitted_at":"2024-08-19T15:27:25Z","title":"ARMADA: Attribute-Based Multimodal Data Augmentation","version":1},"cited_work":{"arxiv_id":"2408.10086","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.10086","snapshot_observed_at":"2026-08-07T14:47:27.699275Z","title":"ARMADA: Attribute-Based Multimodal Data Augmentation","venue":"cs.AI","work_id":"3e5693df-934d-4760-b4d7-9688523d2832","year":2024},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:21.743339Z"},"links":{"cited_paper":"/paper/2408.10086","citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:8188ebb35da1701022abf9c477fa76b515abec2360852ebcb8b56a6b199b721e","observation_id":"04aa2b42-bb28-45f9-991e-09beeba581d7","resolution":{"observed_at":"2026-08-07T14:47:27.779282Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:37.730601Z","title":"Referitgame: Referring to objects in pho- tographs of natural scenes","venue":null,"work_id":"307c5fa1-43eb-40f7-9b39-07ffc2118017","year":2014},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:21.838974Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:447a58ec5ec936e5215ae8359913e933e143b8ac79d6f8c2bb57791da8fbab9f","observation_id":"81d70f02-e4d9-499e-a0d0-6180e3306afe","resolution":{"observed_at":"2026-08-07T14:47:37.826592Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:37.497452Z","title":"Segment any- thing","venue":null,"work_id":"67fcfcb0-ac6b-4de1-b020-9ce7be45f241","year":2023},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:21.921379Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:2220db3f33704cc75cc2321b11d8b20e9a08297cfb709eb6a01279661e5b79fe","observation_id":"ae266e83-b34f-4ec4-97a8-e409dbdb1892","resolution":{"observed_at":"2026-08-07T14:47:37.601437Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:37.208877Z","title":"Lisa: Reasoning segmentation via large language model","venue":null,"work_id":"ed9e5410-f4bd-4a1f-8415-08d25013ec93","year":2024},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:22.006435Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:e95425317f7d855c84e20ea961e61a8bded0bf1c5614b53805c7f62e95d5b138","observation_id":"1bacd787-1def-49b8-b1b3-46cb72466998","resolution":{"observed_at":"2026-08-07T14:47:37.363436Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:36.953305Z","title":"Bigdatasetgan: Synthe- sizing imagenet with pixel-wise annotations","venue":null,"work_id":"47cf7577-8635-439c-9dda-0355d515b342","year":2022},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:22.120552Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:33d32aba4312e9ec840b7856cd1e2e2934cfcc5bb07f8e42882561c2e6682302","observation_id":"698c7ee5-b400-4e3a-9262-594cd968a8ac","resolution":{"observed_at":"2026-08-07T14:47:37.057411Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:36.733177Z","title":"Gres: Gener- alized referring expression segmentation","venue":null,"work_id":"798c458f-f8b4-49bb-af02-0d99c3c975e9","year":2023},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:22.239095Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:bd5bd76c25113f049fd8716017793f63ba21a90fd440a3bab3e0ce33cf681f11","observation_id":"b7c1f8cc-3b97-41dd-b890-b773d04e95ec","resolution":{"observed_at":"2026-08-07T14:47:36.829544Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:36.489300Z","title":"Improved baselines with visual instruction tuning","venue":null,"work_id":"1fe37c8c-c332-4eb1-a0fb-b44a2ecda485","year":2024},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:22.322938Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:10eb67643084938efb27f0c1e1835c4044b5e7b73973f5435d2ca343866ac314","observation_id":"f3e6ca16-c619-4cac-8b4d-c154571f153c","resolution":{"observed_at":"2026-08-07T14:47:36.591923Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:36.250363Z","title":"Visual instruction tuning","venue":null,"work_id":"381f82a4-9954-4de7-9f24-9b1c66d7b403","year":2024},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:22.405143Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:daae25ce9826bac02d829d200aae89fe9702724a76e499697294939a17552f8a","observation_id":"e1829636-215f-4de0-8a45-31ebf64feeab","resolution":{"observed_at":"2026-08-07T14:47:36.367977Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:35.981959Z","title":"Learning multimodal data augmentation in feature space","venue":null,"work_id":"ca34b59f-af25-45ae-871b-d748ce19d70f","year":2023},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:22.497171Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:4d2845f49fd1fb090f5a69b7fdeafeeb7e027b04dfbab5eaca4ea53dbf894142","observation_id":"7b0b0c59-ebbc-4ddc-bb91-8141aea95786","resolution":{"observed_at":"2026-08-07T14:47:36.085520Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:35.621367Z","title":"Generation and com- prehension of unambiguous object descriptions","venue":null,"work_id":"6999331f-867b-4958-9488-aa38c69cbe5c","year":2016},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:22.582842Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:27dd51f668856ccf661a8ca9025e725448cd832e569987d83239a224714bda55","observation_id":"2b756d41-74bb-40b7-8fa0-0565df969ac0","resolution":{"observed_at":"2026-08-07T14:47:35.769701Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:35.375128Z","title":"Arm- bench: An object-centric benchmark dataset for robotic ma- nipulation","venue":null,"work_id":"29883de5-3c2f-4945-97d4-f171491c486b","year":2023},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:22.655738Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:b8b000856f78b53de503928f83e594389e61d9ada36c6f0172a7d751de286eb7","observation_id":"2d0486e6-a933-42fd-a5ad-faba589ad9f0","resolution":{"observed_at":"2026-08-07T14:47:35.465501Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:35.111647Z","title":"The mapillary vistas dataset for semantic understanding of street scenes","venue":null,"work_id":"983f3a66-7956-46d2-b5e3-242c956863b9","year":null},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:22.831522Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:5e5e45e871200b42c0b4de95abde34a7557e355e47148397d24f1869471157f7","observation_id":"376cf2ff-cf1e-4986-b3dc-81d7c8a7364a","resolution":{"observed_at":"2026-08-07T14:47:35.253944Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:34.812467Z","title":"Dataset diffusion: Diffusion-based synthetic data generation for pixel-level semantic segmentation","venue":null,"work_id":"bf0ec2a9-e27d-497f-8d01-d29cf7885ecd","year":2023},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:22.906508Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:5c4421e833709ec422d3515572beb3576cec783f1187afdd74c72a2aa6290911","observation_id":"8c7a92a1-16bf-4206-a287-7634dca742da","resolution":{"observed_at":"2026-08-07T14:47:34.982424Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:34.569940Z","title":"Learning transferable visual models from natural language supervision","venue":null,"work_id":"75a9c6f8-e9c7-4cc6-acc7-8917e430a2ef","year":2021},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:23.020067Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:4a584b877a35f4384f7830f54d0ac3395e7cc778a632f9e3e886cf5ffc0949dc","observation_id":"6e601524-4404-4457-948b-7ebddc73b350","resolution":{"observed_at":"2026-08-07T14:47:34.697135Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:34.235391Z","title":"Paco: Parts and attributes of common objects","venue":null,"work_id":"47e10bbf-9008-4f0d-8258-15b33a577e0e","year":null},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:23.143071Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:8c9eb9d35f54c5c61be9ae134f81bd756d35080b2ccdd794ceb91d182f3ac336","observation_id":"e2e5d3db-c511-4199-8a54-e25671517fc1","resolution":{"observed_at":"2026-08-07T14:47:34.330678Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:33.921262Z","title":"Glamm: Pixel grounding large multimodal model","venue":null,"work_id":"8fbe57ad-4363-45ec-91a5-1c280531d20c","year":2024},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:23.247935Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:e8f724ce3e75e718f6579d538268a98b1aedf6bceae0f816d8d7d54568409458","observation_id":"fa6d6adf-7a56-43be-8858-94a792959ffd","resolution":{"observed_at":"2026-08-07T14:47:34.086363Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:33.602571Z","title":"SAM 2: Segment anything in images and videos","venue":null,"work_id":"9a2869d3-9d63-4c53-a46a-40de4e84f0bf","year":null},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:23.370944Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:5c83a2ec96102a5188b2bfd750bd088e3de106569e4ba9da5a37582c3b426bec","observation_id":"86e7719c-f194-4a07-ab95-adc4a619e2ed","resolution":{"observed_at":"2026-08-07T14:47:33.780743Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:33.381905Z","title":"Pixellm: Pixel reasoning with large multimodal model","venue":null,"work_id":"ba7d32b0-91e5-415f-b93c-43fae616bd17","year":2024},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:23.476000Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:07cf5506a0ad48cf61bed882e7248f50f507d10b292ee3143a8577cfabc87068","observation_id":"c4b4b0f9-a4a7-4e5d-bbd5-950cb452100a","resolution":{"observed_at":"2026-08-07T14:47:33.485800Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:33.135262Z","title":"Grounding of textual phrases in images by reconstruction","venue":null,"work_id":"42cd932a-f0fa-493c-9a0d-e28b3b86fb9e","year":null},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:23.585827Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:494fe5c29639043a3abde6e25a957f8f9084d871be142ae73ba73f8d206b409c","observation_id":"9cb64232-9601-4aaa-8e40-c94bc1fd775a","resolution":{"observed_at":"2026-08-07T14:47:33.238447Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1805.00123","last_updated":"2018-04-30T22:49:54Z","snapshot_observed_at":"2026-08-14T19:20:17.924747Z","submitted_at":"2018-04-30T22:49:54Z","title":"CrowdHuman: A Benchmark for Detecting Human in a Crowd","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1805.00123","snapshot_observed_at":"2026-08-07T14:47:23.699161Z","title":"Crowdhuman: A bench- mark for detecting human in a crowd","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:23.699161Z"},"links":{"cited_paper":"/paper/1805.00123","citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:36aba6499b2bd5a963e5601d25af11ab42fcf4d2fb9458102e25636e3e5b9808","observation_id":"a7751b07-0325-4461-8730-f2f0486faf37","resolution":{"observed_at":"2026-08-07T14:47:23.699161Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:32.909721Z","title":"Denoising diffusion implicit models","venue":null,"work_id":"afda224a-8d74-4953-af9b-531e7afb4742","year":2021},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:23.808606Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:0f2076c9276ff808dcc75134b64e019121ff5c921716918750e63bf91695ff03","observation_id":"a09e70b1-3ba4-4121-a076-ae22b72a40aa","resolution":{"observed_at":"2026-08-07T14:47:33.015018Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.02048","last_updated":"2025-05-28T08:11:16Z","snapshot_observed_at":"2026-08-16T12:59:45.220057Z","submitted_at":"2025-01-03T19:00:00Z","title":"DreamMask: Boosting Open-vocabulary Panoptic Segmentation with Synthetic Data","version":2},"cited_work":{"arxiv_id":"2501.02048","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.02048","snapshot_observed_at":"2026-08-07T14:47:27.478498Z","title":"DreamMask: Boosting Open-vocabulary Panoptic Segmentation with Synthetic Data","venue":"cs.CV","work_id":"98fc4bb5-4763-48b9-8b16-6c2dee21daf2","year":2025},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:23.916439Z"},"links":{"cited_paper":"/paper/2501.02048","citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:0b05a566294f9f401b20d1909baa76fc47f70f1cc07bc93b764e4cfd878be662","observation_id":"5f946cb7-fbce-46cf-8317-c2e35efda692","resolution":{"observed_at":"2026-08-07T14:47:27.583189Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:32.636223Z","title":"Cris: Clip-driven referring image segmentation","venue":null,"work_id":"7cb59edd-5241-419b-874b-8301b93a9a85","year":null},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:24.069836Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:3fb1cb0eca75f36b0273f75603fc8f91edab3fecd3fb030ccfdd8d6622973c8b","observation_id":"110b8971-dade-48ef-83e0-968aa9e3a808","resolution":{"observed_at":"2026-08-07T14:47:32.788625Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01330","last_updated":"2023-10-02T16:48:50Z","snapshot_observed_at":"2026-08-16T14:55:48.344596Z","submitted_at":"2023-10-02T16:48:50Z","title":"Towards reporting bias in visual-language datasets: bimodal augmentation by decoupling object-attribute association","version":1},"cited_work":{"arxiv_id":"2310.01330","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.01330","snapshot_observed_at":"2026-08-07T14:47:27.261656Z","title":"Towards reporting bias in visual-language datasets: bimodal augmentation by decoupling object-attribute association","venue":"cs.CV","work_id":"e10fd36f-9ce4-4e6b-bb0f-a4130b72b569","year":2023},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:24.223505Z"},"links":{"cited_paper":"/paper/2310.01330","citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:3ab304628041e894835f5bcaa9739b9bd9cb6dbf080dbe550c6a2f50a3becc6c","observation_id":"af2dc787-bdba-49e3-aa9e-79ca09ac039e","resolution":{"observed_at":"2026-08-07T14:47:27.366597Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:32.399025Z","title":"Diffumask: Synthesizing images with pixel-level annotations for semantic segmentation using diffu- sion models","venue":null,"work_id":"56165b03-9266-4782-8ef6-df04ef90a59e","year":2023},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:24.377245Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:40a75c330acf598a979ac07f492ec57525dba97555d78947f0053bb970551a44","observation_id":"311c22a3-2dc1-427d-811d-a8c2a2111c5b","resolution":{"observed_at":"2026-08-07T14:47:32.549960Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:32.169723Z","title":"Gsva: Generalized segmentation via multimodal large language models","venue":null,"work_id":"189931f4-0e13-42e1-9032-81ca4ef8a268","year":2024},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:24.527351Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:3a983f6d62bde5780806dff908672a940109b882e49e52f54c03ef560d7c9554","observation_id":"65cc8935-735d-46fc-a756-96c111073256","resolution":{"observed_at":"2026-08-07T14:47:32.290075Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.10629","last_updated":"2024-10-20T14:35:31Z","snapshot_observed_at":"2026-08-16T01:17:28.297637Z","submitted_at":"2024-10-14T15:36:42Z","title":"SANA: Efficient High-Resolution Image Synthesis with Linear Diffusion Transformers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.10629","snapshot_observed_at":"2026-08-07T14:47:24.674218Z","title":"Sana: Efficient high-resolution image synthesis with lin- ear diffusion transformers","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:24.674218Z"},"links":{"cited_paper":"/paper/2410.10629","citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:e9144f81f9479ac5ce61b099fbd00a1643417c18a54622c1740d654053e78533","observation_id":"664af2dd-7586-4e2f-b56f-a4755bf3bebb","resolution":{"observed_at":"2026-08-07T14:47:24.674218Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:31.960808Z","title":"Mosaicfusion: Diffusion models as data augmenters for large vocabulary instance segmentation","venue":null,"work_id":"08bee63e-2ba2-4cf0-bb72-fba12dc841b9","year":2024},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:24.804534Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:ae6b69f3dd817fad11d294d380ef1abf7b5343d1f264fc05a615fc7ba39435a0","observation_id":"f9c2fbe1-7532-40b5-a8b8-a64bd7d7408e","resolution":{"observed_at":"2026-08-07T14:47:32.060147Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:31.732033Z","title":"Bridging vision and language encoders: Parameter-efficient tuning for referring image segmentation","venue":null,"work_id":"3a8b7254-c13e-42ac-8e0a-1189616da876","year":2023},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:24.935045Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:63cf69dc9be558061e1e8c71fbb1dfd5295eda386eec9728f0d9a0579f2b122b","observation_id":"896b32b1-bbf4-4502-85ab-79dc11b0e4f7","resolution":{"observed_at":"2026-08-07T14:47:31.877017Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:31.466874Z","title":"Panoptic scene graph gen- eration","venue":null,"work_id":"6b88279d-ee3c-41a7-9c30-72941647e486","year":2022},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:25.078638Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:4639b952cf7d6407564d8c1b3e72ffe79f34e577bb647363c273ef6c037b8805","observation_id":"d107d3a5-db7d-4cc3-b2c0-5f562362d384","resolution":{"observed_at":"2026-08-07T14:47:31.620806Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:31.182229Z","title":"Freemask: Synthetic images with dense annotations make stronger segmentation models","venue":null,"work_id":"239f77d1-7a2c-4639-81b7-5cf90a1429a4","year":null},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:25.211282Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:9f36b18c8146a7e525c7f7dc8a5bc4a3a4d9cfaad2867327a7f459c4040e4cb9","observation_id":"1365327d-86c0-4d4a-b647-c82600da3a52","resolution":{"observed_at":"2026-08-07T14:47:31.349947Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:30.896671Z","title":"Lavt: Language-aware vi- sion transformer for referring image segmentation","venue":null,"work_id":"75e0ea6e-8de6-427e-8130-af11c8cdb849","year":2022},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:25.303573Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:fd0b4ad71021594ff7fcc8cff8e3b80abc5bdfcb2d85fac3b862c698f0896cce","observation_id":"b8f52a39-dab2-41d0-8719-3e2d99ac9dbc","resolution":{"observed_at":"2026-08-07T14:47:31.049158Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:30.642365Z","title":"Seggen: Supercharging segmentation models with text2mask and mask2img synthesis","venue":null,"work_id":"caba22cf-16d4-4132-85bb-4d4306b0d527","year":2024},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:25.460160Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:46440b36d33f42bfd0fbdfccbc2e871aadf31e84962f7c2b87991b4973d8b3c1","observation_id":"5b90db18-c264-45a0-a000-dc216eddebb7","resolution":{"observed_at":"2026-08-07T14:47:30.759804Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13893","last_updated":"2025-01-23T18:08:57Z","snapshot_observed_at":"2026-08-13T23:14:19.264762Z","submitted_at":"2025-01-23T18:08:57Z","title":"Pix2Cap-COCO: Advancing Visual Comprehension via Pixel-Level Captioning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13893","snapshot_observed_at":"2026-08-07T14:47:25.624443Z","title":"Pix2cap-coco: Advancing visual comprehension via pixel-level captioning","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:25.624443Z"},"links":{"cited_paper":"/paper/2501.13893","citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:0e880e5a6e423af66f831489708c73be89314efcfabf3392aa00b9e8eafea7e2","observation_id":"9e21ba7b-94d9-4cd1-8b4c-13e978fbf85c","resolution":{"observed_at":"2026-08-07T14:47:25.624443Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:30.395767Z","title":"From image descriptions to visual denotations: New similarity metrics for semantic inference over event descrip- tions","venue":null,"work_id":"6177647f-ed69-404d-9d7b-65f270573dc1","year":2014},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:25.738988Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:a38652c4ae85f0ef4dd155176b67c865b00481eebafb5fef4df18b8e8cb52d77","observation_id":"be04d225-2071-4d7c-9095-03a4d872127f","resolution":{"observed_at":"2026-08-07T14:47:30.484684Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:30.140045Z","title":"Coca: Contrastive captioners are image-text foundation models","venue":null,"work_id":"b69b49da-a509-42cf-a38a-5a604d740f54","year":2022},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:25.851673Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:acb32da382e1324b8fb79266ffbab6341229b19508adb16741da892c16e2f7bd","observation_id":"acee422e-2b11-4869-b89b-d5faaecd52b9","resolution":{"observed_at":"2026-08-07T14:47:30.246850Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:29.823972Z","title":"Modeling context in referring expressions","venue":null,"work_id":"859f06ad-be0e-4b03-bc9a-b153556b55ff","year":2016},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:25.960341Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:eec647fdbe3e41a851cc7ff71259254c272a585d6bb320b09a89e9cc5d73a6b3","observation_id":"298725ce-b1b7-465e-9404-4ce1c629a6b5","resolution":{"observed_at":"2026-08-07T14:47:30.004079Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:29.523382Z","title":"Pseudo- ris: Distinctive pseudo-supervision generation for referring image segmentation","venue":null,"work_id":"0b19d853-3fad-49e1-a96d-78ac80921515","year":2024},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:26.080046Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:ca313d2324f59cd6ef57dd8362530dd95ef1f8fc73a106648c0e6d4b600adefa","observation_id":"53ee01ac-5595-4c72-ad2d-3706a951b1f9","resolution":{"observed_at":"2026-08-07T14:47:29.677320Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:29.272982Z","title":"Revisiting counterfactual prob- lems in referring expression comprehension","venue":null,"work_id":"d682486e-39a8-4fcd-a97e-5501178e06bd","year":2024},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:26.169375Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:07245502244041dcd7ddb1f933baf2cd3575e915f27cf6f2139c373248e3f574","observation_id":"029d00c9-a3b2-46ca-a571-56dae9d18660","resolution":{"observed_at":"2026-08-07T14:47:29.418166Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:29.062602Z","title":"Datasetgan: Efficient labeled data factory with minimal human effort","venue":null,"work_id":"df2ae53e-1c1b-47e6-956e-b080396f8224","year":2021},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:26.285795Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:9eb366fbcafe1179467884cecf858ee6b31e7a9e2c909ef66935a7b997dcdcea","observation_id":"f413083f-ea99-4897-bd65-4ed33e1fddfe","resolution":{"observed_at":"2026-08-07T14:47:29.165630Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.20076","last_updated":"2025-03-10T12:34:24Z","snapshot_observed_at":"2026-08-16T13:38:39.890570Z","submitted_at":"2024-06-28T17:38:18Z","title":"EVF-SAM: Early Vision-Language Fusion for Text-Prompted Segment Anything Model","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.20076","snapshot_observed_at":"2026-08-07T14:47:26.416689Z","title":"Evf-sam: Early vision-language fusion for text-prompted seg- ment anything model","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:26.416689Z"},"links":{"cited_paper":"/paper/2406.20076","citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:bf589d7a974616dc405b17a452e61f9aadbc26caac099724a5d81dcffaa0ddff","observation_id":"eea956fa-8b55-47b8-904c-11aabd935bef","resolution":{"observed_at":"2026-08-07T14:47:26.416689Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:28.801476Z","title":"Psalm: Pixelwise segmentation with large multi-modal model","venue":null,"work_id":"d4c6cb6b-3f5b-4d80-b4e4-60c2d6516b50","year":2024},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:26.551994Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:5f9f505832bbab85a6b0c689ba86df122ba9b86dbe941c8eb059f711a823abc7","observation_id":"ca4ff209-3265-47d3-8c01-6989a655964e","resolution":{"observed_at":"2026-08-07T14:47:28.928964Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:28.588830Z","title":"X-paste: Revisiting scalable copy-paste for in- stance segmentation using clip and stablediffusion","venue":null,"work_id":"524c8727-0cb6-46e2-8fed-5daa773a5833","year":2023},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:26.658210Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:064b9897aadf472398c6b7366d0eebfd16c445775e2a85783187b7c1938420f4","observation_id":"f7082026-268a-4aff-b4bf-f8b162d5061d","resolution":{"observed_at":"2026-08-07T14:47:28.685348Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:26.760787Z","title":"Unleashing text-to-image diffusion models for visual perception","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:26.760787Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:4a336d91772968697c05bffef19843e2b42129649c9068d6c031d7b3685dd244","observation_id":"faaab8c1-8e33-4b9d-bde0-3cda016368a8","resolution":{"observed_at":"2026-08-07T14:47:26.760787Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:28.368649Z","title":"Scene parsing through ade20k dataset","venue":null,"work_id":"3a25d8c5-c91f-4fb6-9429-257f2cc012ef","year":2017},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:26.890631Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:09c04d7f928d3234bef1c7322b1b6e32bd6469edc9095d6e3cde16a08c456a21","observation_id":"7a5485ab-0c7b-4bbc-a4f3-e759d859ec47","resolution":{"observed_at":"2026-08-07T14:47:28.461809Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:28.071759Z","title":"Generalized decoding for pixel, image, and language","venue":null,"work_id":"c28438d4-a8f8-45ab-8e82-ccbee2efa3c3","year":2023},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:27.016843Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:5fead3515c9017863666693bb7c009f47c6c2b88c0bfbe0c77c7e1bd262feb04","observation_id":"a9a2d513-693c-4783-b193-ac6be2c89da3","resolution":{"observed_at":"2026-08-07T14:47:28.237992Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:27.880708Z","title":"the cat sitting on the bench next to big green wooden boat in the center of the image","venue":null,"work_id":"dc5938a9-f528-4ead-a305-4ff41e0dcd53","year":2023},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:27.120881Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:3899f2a91854ad2ff75463f49b3da3571cbcced11050efe2dd3411370cc588fa","observation_id":"926b136e-3d50-499b-99db-ad42737ca299","resolution":{"observed_at":"2026-08-07T14:47:27.953849Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-13T20:14:44.136505Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data"},"reference_resolution":{"displayed":60,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":8,"verified_exact":3,"verified_fuzzy":49},"total_outbound_references":60},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"thesis":"As of 17 August 2026, this Paper Citation Record lists 60 of 60 outbound references and 0 inbound Pith citation observations for arXiv:2505.17695."}