{"as_of":"2026-08-10T10:41:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:42bc7902c2d3e144c25c9a6c84de737af86992da1e141a010e34aa36921f5135","coverage":[{"denominator":68,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":68,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T19:23:15.210670Z","state":"measured"},{"denominator":68,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":68,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2507.05798/citation-record","integrity":"/paper/2507.05798/integrity","json":"/paper/2507.05798/citation-record.json","paper":"/paper/2507.05798"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2211.01324","last_updated":"2023-03-14T00:22:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-11-02T17:43:04Z","title":"eDiff-I: Text-to-Image Diffusion Models with an Ensemble of Expert Denoisers","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2211.01324","snapshot_observed_at":"2026-08-06T19:23:13.558538Z","title":"ediff-i: Text-to-image diffusion models with an ensem- ble of expert denoisers","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:13.558538Z"},"links":{"cited_paper":"/paper/2211.01324","citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:9a55385f35aed58e993a3587915bc090781704d57e8c8805c33971065bec068c","observation_id":"e6c1f154-424c-4e6e-8c97-0885f4d72d77","resolution":{"observed_at":"2026-08-06T19:23:13.558538Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.433751Z","title":"Hico: A benchmark for recognizing human-object interactions in images","venue":null,"work_id":"fcc197fe-5a9b-4d42-98f7-a467c2dffc06","year":null},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:13.604400Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:4933d08eec20b52651ad02d1998edfb629651afb402b5c2a7d883ee3439d8900","observation_id":"a5cbe5f6-98ba-4344-ab0f-838db8bbb4c1","resolution":{"observed_at":"2026-08-06T19:23:16.438037Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.420476Z","title":"Spatialvlm: Endow- ing vision-language models with spatial reasoning capabili- ties","venue":null,"work_id":"16b6c266-e0b7-461d-81d0-dff0e7dcd8d1","year":2024},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:13.778985Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:be1272a1222be575d5ff641b6f9da280e2428037ba90d4f6ac237b88da0d1aa4","observation_id":"005633a3-85e2-4f05-9326-f3d7e62cff7a","resolution":{"observed_at":"2026-08-06T19:23:16.424545Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.407411Z","title":"Expanding scene graph boundaries: fully open-vocabulary scene graph generation via visual-concept alignment and retention","venue":null,"work_id":"b5a5a840-91a0-45a6-8f14-8540fa37c2d8","year":2024},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:13.901264Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:078fcecb420261463bfd5d5d500c59d3dbc606e2ec67737063f65da1990fff0d","observation_id":"1be1c4b7-8b93-4eb6-8d1f-1496ce0c38d8","resolution":{"observed_at":"2026-08-06T19:23:16.411541Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.394191Z","title":"Spatial- rgpt: Grounded spatial reasoning in vision-language mod- els","venue":null,"work_id":"90f23ca9-7204-47ea-ba22-73a144a7f027","year":2025},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:14.000753Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:bcc8a8c0c55219047735feef1be0fb519c4c3e33c4b8c6119526f1c5b972e5a4","observation_id":"fcc6fb54-c3fd-4be6-8248-ccb32eaadad3","resolution":{"observed_at":"2026-08-06T19:23:16.398275Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.380731Z","title":"Masked-attention mask transformer for universal image segmentation","venue":null,"work_id":"f0e82e78-d8c7-470e-879e-158ae1acf4c4","year":2022},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:14.118018Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:909893cef7ca97ba431a1521b4222cd58713b430adc008a273423876c51d121b","observation_id":"4eb96ac1-1921-4d84-9f32-c7d91bdfb551","resolution":{"observed_at":"2026-08-06T19:23:16.384940Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.367133Z","title":"Recovering the unbiased scene graphs from the biased ones","venue":null,"work_id":"fdb6820d-32c7-4f6b-a6fe-ea2de5a0de6a","year":2021},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:14.227404Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:1737bc71bc96ab63eb87b7f6698fd29b555224667dfbde09fbf282e0b52d5dbe","observation_id":"dcd6c2bc-d51c-4c24-9184-d58110c75fd2","resolution":{"observed_at":"2026-08-06T19:23:16.371152Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.353589Z","title":"Reltr: Relation transformer for scene graph generation.IEEE Transactions on Pattern Analysis and Machine Intelligence, 45(9):11169–11183, 2023","venue":null,"work_id":"36d841bf-72f7-4779-a9bc-159640a635d8","year":2023},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:14.387287Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:f9439fdf60a1c054c83f7c63253ce090adcf315e21fe1eb3731fb2dd37cfd3ef","observation_id":"15167482-b84b-4d8e-85d4-87a7163c1d0b","resolution":{"observed_at":"2026-08-06T19:23:16.358031Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.339685Z","title":"De- coupling zero-shot semantic segmentation","venue":null,"work_id":"9df1f48f-a0fe-4758-8a46-8f25848fe30d","year":2022},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:14.489106Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:9e407b7bc04b41620302a833141ebe94361a58faee7f78cdf0fdd42392530431","observation_id":"170c9556-717f-43f7-9b4a-3e42ebe5f63b","resolution":{"observed_at":"2026-08-06T19:23:16.344306Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.326359Z","title":"Concept sliders: Lora adaptors for precise control in diffusion models","venue":null,"work_id":"9e21fb2e-6b96-470a-b338-47574e0ed503","year":2024},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:14.635239Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:48882ad9d7040bc1721c43093e13d08055fd9d3c4f17536ab8f1d87d5fa010d7","observation_id":"6ee591b2-59ea-4bf1-b39c-3338f986638d","resolution":{"observed_at":"2026-08-06T19:23:16.330702Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.312049Z","title":"ican: Instance- centric attention network for human-object interaction detec- tion","venue":null,"work_id":"c86faf57-2fea-447b-bb7f-e91b828aef4b","year":2018},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:14.761753Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:84176a541a79e319f67c35605c1174c597849ad3f70c4efb4641cd6e467536f5","observation_id":"0844e64d-0997-4aee-84f6-eafa729a12c5","resolution":{"observed_at":"2026-08-06T19:23:16.316366Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.298814Z","title":"Open-vocabulary object detection via vision and language knowledge distillation","venue":null,"work_id":"067a1916-369e-48f1-baf0-3d73cf6d7141","year":2022},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:14.844366Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:e4f04b237fd3f324b523b53fe2843479a8b5961c757a446ef49c44da86836e71","observation_id":"38bedc89-6f44-488e-9072-617542aaa39a","resolution":{"observed_at":"2026-08-06T19:23:16.302822Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.285092Z","title":"Dsgg: Dense relation transformer for an end-to-end scene graph generation","venue":null,"work_id":"a3d53948-9bca-4ba1-91d9-89798deba29c","year":2024},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:14.973198Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:3c6f8a80cec628551af09ff322c3380b85a6dfe4867badf9e0c3c54c2c1ac833","observation_id":"4a89acf0-82f2-42ad-bbd9-b33ea4c630ae","resolution":{"observed_at":"2026-08-06T19:23:16.289314Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.271013Z","title":"Learning from the scene and borrowing from the rich: tackling the long tail in scene graph generation","venue":null,"work_id":"87e11dc0-a090-4cf5-a0d0-057698a67952","year":2021},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:14.977613Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:f4965139590f46a28473f73e0740f99886444f8076054aa69617933f6b0231c2","observation_id":"ce04be5c-e975-4890-9ee2-b5cd047c104a","resolution":{"observed_at":"2026-08-06T19:23:16.275295Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2108.08600","last_updated":"2021-08-19T10:13:55Z","snapshot_observed_at":"2026-07-06T11:39:34.310174Z","submitted_at":"2021-08-19T10:13:55Z","title":"Semantic Compositional Learning for Low-shot Scene Graph Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2108.08600","snapshot_observed_at":"2026-08-06T19:23:14.982119Z","title":"Semantic compositional learning for low-shot scene graph generation","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:14.982119Z"},"links":{"cited_paper":"/paper/2108.08600","citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:5df1c3f3e7cd667a61607133b8e367be335284b0ef694660623ae33bc7bbc029","observation_id":"d319bbd3-3677-4553-bfeb-b1b40f6c5861","resolution":{"observed_at":"2026-08-06T19:23:14.982119Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.256401Z","title":"Ex- ploiting scene graphs for human-object interaction detection","venue":null,"work_id":"526afcd1-228f-417a-a9bc-a6ff01d437b2","year":2021},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:14.986482Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:b65e5275e6b59bfe4ee4fb4ddb309f480fbeed972c269091a89550176ea34b3e","observation_id":"df8ae0da-2210-4c84-bacf-2e746426e7f9","resolution":{"observed_at":"2026-08-06T19:23:16.260984Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.242241Z","title":"To- wards open-vocabulary scene graph generation with prompt- based finetuning","venue":null,"work_id":"76b7e5d9-f66a-4945-b313-d96f10c5b44b","year":2022},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:14.990928Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:641b07b1e8f832c087a57daf07dd3a46fc84885d67c918cefd688a1bd4b3409e","observation_id":"5d1dcdcb-0adf-417e-9c2f-a059f2ca0c83","resolution":{"observed_at":"2026-08-06T19:23:16.246648Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.226860Z","title":"To- ward a unified transformer-based framework for scene graph generation and human-object interaction detection","venue":null,"work_id":"70fc2203-8c75-4006-a0e1-022afe39d84a","year":2023},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:14.994859Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:5df36767699b65d204bec2f3798fd5e842af239e5f395c2431c4fa80f24b391c","observation_id":"b82d42b7-4d1e-43f6-a206-0a28444edef6","resolution":{"observed_at":"2026-08-06T19:23:16.231947Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.212419Z","title":"Hu, Yelong Shen, Phillip Wallis, Zeyuan Allen- Zhu, Yuanzhi Li, Shean Wang, Lu Wang, and Weizhu Chen","venue":null,"work_id":"07384bb2-8769-4919-814b-46f2e64ea621","year":2022},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:14.998881Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:6b5fa45961492059417bff2441c0d0bbe0383def339b761c04eb8eac86a0ac58","observation_id":"06435c4a-dfbd-4a64-b90e-c4636381af1a","resolution":{"observed_at":"2026-08-06T19:23:16.216462Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.196925Z","title":"Egtr: Extracting graph from trans- former for scene graph generation","venue":null,"work_id":"90a7225b-495a-49fe-bbfd-7b13ae73006b","year":2024},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.003205Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:fc23bac10474cbe6f9612fd814e28db819ecf945abb1ca9787b642584ed3ee5a","observation_id":"da90bef4-3968-4ad9-87c7-bd06190c5f16","resolution":{"observed_at":"2026-08-06T19:23:16.201664Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.181531Z","title":"Zero-shot scene graph relation prediction through commonsense knowledge inte- gration","venue":null,"work_id":"7d8c4d47-6956-4e2c-ab54-e0472507fcc1","year":2021},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.007201Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:7ed8426538db2c68b273ea59a9d583ebce9321c0f4079544d628098f098f8b44","observation_id":"37443cb2-411b-4f47-ad8f-a629003eb7e4","resolution":{"observed_at":"2026-08-06T19:23:16.185701Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.167096Z","title":"Imagic: Text-based real image editing with diffusion models","venue":null,"work_id":"eb22562a-11f3-4168-aec0-0873cac5bd78","year":2023},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.011704Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:85f41fbed04597bf4925c3dd5b04b21185389fe67d9220c380c04e0e6d59e4ec","observation_id":"9dd773e6-a11b-4497-bf4a-7a5cf6b03f21","resolution":{"observed_at":"2026-08-06T19:23:16.171509Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.152392Z","title":"Text-image align- ment for diffusion-based perception","venue":null,"work_id":"e66a3070-a2df-4624-bc1e-a97112ad2e0f","year":2024},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.015861Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:c9d69ea10ef1721f3e14477ed4a512bcee6c5ef093757c9dfcb5038021904b71","observation_id":"68dc0939-d405-42e9-8f93-6b34c55aace4","resolution":{"observed_at":"2026-08-06T19:23:16.156666Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.138742Z","title":"Visual genome: Connecting language and vision using crowdsourced dense image annotations","venue":null,"work_id":"022e8fe4-4ef4-4e72-b723-d3711b81accd","year":2017},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.019691Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:6a80db7bd58703ba9d6faba01cdd6b1a41c5170a28bebaf94b5edf1811cf755f","observation_id":"be956cd2-2ca7-478d-b0fc-fc2c0d07aadc","resolution":{"observed_at":"2026-08-06T19:23:16.143198Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.123778Z","title":"Topviewrs: Vision-language models as top-view spatial reasoners","venue":null,"work_id":"9239f237-efd8-41a5-ba13-792da251cb5f","year":2024},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.024093Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:3d9f9c9cd28e438c3f32951c059c44c5495ad6e8ec88779108d632f14d464fe7","observation_id":"fe5635cc-04e6-4064-8929-226ff622c790","resolution":{"observed_at":"2026-08-06T19:23:16.128466Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.028859Z","title":"Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.028859Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:4af7046ed85e9a83298347fc7edad7140579b5f978a5fdd5d133f45572194d33","observation_id":"f046e346-5bd2-492f-a536-fe5dc20f3de0","resolution":{"observed_at":"2026-08-06T19:23:15.028859Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.099181Z","title":"Panoptic scene graph genera- tion with semantics-prototype learning","venue":null,"work_id":"19d93e1f-86bb-4969-baf6-464f89b014a4","year":2024},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.033359Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:14e1b3fcb6681ce8d0c5258e947110abed40d5f70c65afd65bf9483786e5be7c","observation_id":"eae83f39-788f-492c-8ddc-98fabd4a1e67","resolution":{"observed_at":"2026-08-06T19:23:16.103340Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.042930Z","title":"Grounded language-image pre-training","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.042930Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:ff34a77f31f829bce99be9ec2005affd1979fa3e26d30f6b870bc77f640f779f","observation_id":"0ffe0184-9e90-4a6e-a8ee-f0e8476d4c99","resolution":{"observed_at":"2026-08-06T19:23:15.042930Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.074952Z","title":"Sgtr: End- to-end scene graph generation with transformer","venue":null,"work_id":"603353b1-2ad1-4081-a178-fbd2d3ffca9f","year":2022},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.047144Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:5f2c879ca61ab96024c06f4c62017d33c1ff988705f66d6d2e529f425c91985e","observation_id":"1aac3644-1b55-4285-a068-f9eb0300f70c","resolution":{"observed_at":"2026-08-06T19:23:16.078982Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.059601Z","title":"From pixels to graphs: Open-vocabulary scene graph generation with vision-language models","venue":null,"work_id":"43106136-fba5-4b3a-825a-0b875a6a387e","year":2024},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.051619Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:c9f9aad79359f0e006f95a8848b93653e18f0f50f5823be0ae0b524cf57ccf44","observation_id":"1ef41c09-9691-4db3-adff-23682b389116","resolution":{"observed_at":"2026-08-06T19:23:16.064671Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.044279Z","title":"Open-vocabulary object segmentation with diffusion models","venue":null,"work_id":"b3b24f46-b963-4dba-87f6-33a1dedbb72f","year":2023},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.056473Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:1f1f50521b93d8c356b611832a2dd752bf6abecd1597892a6b2b45933c6eca32","observation_id":"8e1b958a-2675-4c82-a32e-ad75ea3a3ff0","resolution":{"observed_at":"2026-08-06T19:23:16.048713Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.029040Z","title":"Gps-net: Graph property sensing network for scene graph generation","venue":null,"work_id":"b9a46323-4c18-4232-a1fb-8aa41316681e","year":2020},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.061125Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:5ed0187d01c31bf3b8ef711b37d17ab4164d9f45b24776a8e20619df18616c17","observation_id":"c6925a11-3925-4f35-ae00-d18804f114b3","resolution":{"observed_at":"2026-08-06T19:23:16.033751Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.013696Z","title":"Path aggregation network for instance segmentation","venue":null,"work_id":"aa30b770-d6da-43c0-b14b-4a1e28a347bf","year":2018},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.065424Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:938e2263d3cc7f9aa2d30676420d357df60f652eca82e9ff2adfb2ba07abe5c4","observation_id":"ce718288-c061-4752-8133-902602392a31","resolution":{"observed_at":"2026-08-06T19:23:16.018071Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.997987Z","title":"Swin transformer: Hierarchical vision transformer using shifted windows","venue":null,"work_id":"adc31042-d8a8-4627-a7cc-f917f00de354","year":2021},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.069551Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:50d25ed02208591730e6923944db8b391d4dd825db2731c6cea90b0fd003bb2c","observation_id":"83209c0b-e7ef-4d13-9d91-7ab56c820be4","resolution":{"observed_at":"2026-08-06T19:23:16.002750Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.982964Z","title":"Attending to graph transformers","venue":null,"work_id":"977c3e77-d437-4c63-8161-a48029f6375f","year":2024},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.074156Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:d0ea35ca1271020dd40922c72960bdfb29f18b4e430e4fe3743688a68ebc306a","observation_id":"0a26314b-c230-407e-a187-3ede07ed7e14","resolution":{"observed_at":"2026-08-06T19:23:15.987147Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.078590Z","title":"Learning transferable visual models from natural language supervi- sion","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.078590Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:70064c30b2cae3cc5c0c780dbe4c5d2d70908012ba9319862cc80787390db67b","observation_id":"072be47a-50cb-41f5-add7-b45233eaab1a","resolution":{"observed_at":"2026-08-06T19:23:15.078590Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.956508Z","title":"Faster r-cnn: Towards real-time object detection with region proposal networks","venue":null,"work_id":"25b36b71-5cd4-459a-96c8-cba32a291437","year":2015},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.082879Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:a79d8b3fae71b015d2a72739143dfa930f8fbe871f4a8c27cf595595a4a0b7d3","observation_id":"31b5836a-ae14-47c3-86f5-45832505cff6","resolution":{"observed_at":"2026-08-06T19:23:15.961046Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.941952Z","title":"High-resolution image syn- thesis with latent diffusion models","venue":null,"work_id":"21edcea0-0661-40bc-b60e-17c283161c75","year":2022},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.086847Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:243466b8fea9d3dfab7226d03abaa01e44f99dd08bfe89704a4e59f6bf12779b","observation_id":"14c277fa-c002-4bdd-9e0c-ab4717562e21","resolution":{"observed_at":"2026-08-06T19:23:15.946551Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.090727Z","title":"Photorealistic text-to-image diffusion models with deep language understanding","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.090727Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:2cd6876372b35f6b19d8fd37fcd1a08c87d58cc4dfd8930ffdb868ca4c120c0a","observation_id":"ff1c0245-ef91-4394-9584-336d8bd4d49f","resolution":{"observed_at":"2026-08-06T19:23:15.090727Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.917430Z","title":"Self- attention with relative position representations","venue":null,"work_id":"0dba3fa9-1e47-42f6-aaf8-12429be0a97a","year":2018},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.094653Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:ef249834d683d32b09e3deb822b0527d1b9691d0dac0e28c9976a63ae64eefc7","observation_id":"896e1bdb-a13f-4142-87fc-23a9935d0ea6","resolution":{"observed_at":"2026-08-06T19:23:15.921672Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.098711Z","title":"Graph trans- formers: A survey","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.098711Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:cab49388c9eb2ffc69dadc156bc8013e737bea28f203d61bbfad1b1181f64445","observation_id":"a164fb9f-4311-48e9-8676-66e56d22f493","resolution":{"observed_at":"2026-08-06T19:23:15.098711Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.902434Z","title":"An empirical analysis on spatial reason- ing capabilities of large multimodal models","venue":null,"work_id":"d36233a8-838f-4948-80d6-e06f16828518","year":2024},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.103044Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:8caa1f7a0b471053f412701a41692a2782df57513053503574e88cb414cbcbdc","observation_id":"f0bc5543-965e-46f5-b57c-361a5a8dff09","resolution":{"observed_at":"2026-08-06T19:23:15.907113Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.887640Z","title":"Continual dif- fusion: Continual customization of text-to-image diffusion with c-lora","venue":null,"work_id":"2c385efc-9d95-43ca-a822-c0193ba72034","year":2024},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.107488Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:a40938eb6a79b629db8e5f4efb7804aef116da9c046c9279804c25445283d09d","observation_id":"f309fc31-829f-438e-9785-bfed2adc4990","resolution":{"observed_at":"2026-08-06T19:23:15.892282Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.873428Z","title":"Denois- ing diffusion implicit models","venue":null,"work_id":"6ceab89e-5cc4-4e39-b47d-8681e6c1493f","year":2021},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.111442Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:4adb6afc4926fa671ff4098dad78240dd1724ebda11981a750625d626d8af575","observation_id":"232d6aea-8873-41e9-aa17-615ad04896b7","resolution":{"observed_at":"2026-08-06T19:23:15.877820Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.859111Z","title":"Transformer-based image generation from scene graphs","venue":null,"work_id":"a34780fe-b848-4606-a86b-c095a84129e0","year":2023},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.115379Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:7ba436ae2c6224edbc21b12e7246127efb55de0060077957a8d48f7bcfe8b132","observation_id":"c8929f1b-89a5-4b9a-86aa-4305bfdb4187","resolution":{"observed_at":"2026-08-06T19:23:15.863788Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.843902Z","title":"Reclip: A strong zero-shot baseline for referring expression compre- hension","venue":null,"work_id":"c127dbce-4bcd-49f3-9723-216a4b5f9f3e","year":2022},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.119376Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:464560fdfcf623140109a730cb82418ffd69847f0ee015c220a98d28cdd4e7c9","observation_id":"3bfc6777-82d2-4515-943e-4e90e990b2d6","resolution":{"observed_at":"2026-08-06T19:23:15.848480Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.828773Z","title":"Learning to compose dynamic tree structures for visual contexts","venue":null,"work_id":"5209093e-7247-46de-85a3-f8098b1bd726","year":2019},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.123341Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:78afb8acab8de6aa787aded3cbdb2f3f6624a360817244c62cdcfe8d07102be8","observation_id":"5f93c6bd-4c6c-46dc-823f-ce3185f399f0","resolution":{"observed_at":"2026-08-06T19:23:15.832992Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.815424Z","title":"Structured sparse r-cnn for di- rect scene graph generation","venue":null,"work_id":"b8bd7efe-7c20-420b-b0ef-92f8e332a60c","year":null},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.127626Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:386555e62fddad2a88f7eb2c12688c46f693d9f4f71667904f4f3d7fdf695ddc","observation_id":"239300a6-bd98-44e7-b7ad-464e469cb0f3","resolution":{"observed_at":"2026-08-06T19:23:15.819616Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.801819Z","title":"Belm: Bidirec- tional explicit linear multi-step sampler for exact inversion in diffusion models","venue":null,"work_id":"d09f7668-21c2-4a3d-ae6d-424dd495b037","year":2025},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.131788Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:6e98b314d6e08db095683862f273afa9f7acdef76322f605371273f1e079b1ee","observation_id":"b16ec409-dbd3-43e7-807f-94325e226eb2","resolution":{"observed_at":"2026-08-06T19:23:15.805997Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.788713Z","title":"Pair then relation: Pair-net for panoptic scene graph generation","venue":null,"work_id":"4450a9cb-8d01-4673-afe1-904cb2e1d7ba","year":2024},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.136220Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:7dd27f5a7a2e098dc97fbcd14c5919753807347fc9bad76aeebae78f59abe434","observation_id":"8cac48ea-bfeb-4125-9511-1ac332747907","resolution":{"observed_at":"2026-08-06T19:23:15.792970Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.671386Z","title":"Stylediffusion: Controllable disentangled style transfer via diffusion models","venue":null,"work_id":"664dfb7c-088d-44fb-8d58-6a1661bea0c0","year":2023},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.140503Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:93ed44c306c46218c6f3eaf38c48f4570119c8ad128693e302f2108d9e4f7429","observation_id":"7559965b-bac0-4096-96c5-ede4733a6a10","resolution":{"observed_at":"2026-08-06T19:23:15.676472Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.657593Z","title":"Chief: Clustering with higher-order motifs in big networks","venue":null,"work_id":"75594cb5-4b0d-4e5f-b265-fbab91523d9b","year":2021},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.144674Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:347ba8a5d374673f465625325743b8b1508b6eed4d3a90f4ce91b95e85453a8f","observation_id":"daa59dd5-6b14-4a9e-a843-504e35d0e036","resolution":{"observed_at":"2026-08-06T19:23:15.661663Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.644499Z","title":"Scene graph generation by iterative message passing","venue":null,"work_id":"3d67df6f-f34d-4ef5-b459-313975ee2977","year":2017},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.148524Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:e6edec4cafdcb75b5b2e59c14c118c07e4294e7ab70dd5cc429e5fad9e864ef8","observation_id":"1421948d-c5ef-4e45-a35a-bbe581f6e78c","resolution":{"observed_at":"2026-08-06T19:23:15.648602Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.630886Z","title":"Scene graph generation by iterative message passing","venue":null,"work_id":"38ee48aa-3e74-498c-ae5c-d969caadd16e","year":2017},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.152769Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:c7cb81164af21f2af9ff5a8724b3b887a8e11ddf0fcfa14b7570f4d1cbb3d283","observation_id":"84bbab9b-e4b5-4ad6-83d1-aeebe41eb33e","resolution":{"observed_at":"2026-08-06T19:23:15.634811Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.617930Z","title":"Open-vocabulary panop- tic segmentation with text-to-image diffusion models","venue":null,"work_id":"5a84d3b3-86ab-43c7-9909-5b58ffa06e32","year":2023},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.157189Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:3c5a6e85308f17c06238234cb15bd76f77fd585db13943956220d5f616434544","observation_id":"927e7fc5-ca20-4a45-9c83-0bf07213b29c","resolution":{"observed_at":"2026-08-06T19:23:15.622102Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.603625Z","title":"Panoptic scene graph gen- eration","venue":null,"work_id":"fecf6476-261b-466b-8f92-2e8cac829258","year":2022},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.161215Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:46b54b495e029198da4454e760bd1cfcb4e8b00f17248a1360a1702d77e1f8f8","observation_id":"1fc9baeb-56e2-4b10-a947-0abcfd5dacb1","resolution":{"observed_at":"2026-08-06T19:23:15.608251Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.590492Z","title":"Open-world human-object interaction detection via multi-modal prompts","venue":null,"work_id":"953141d2-71a1-44ef-9ad5-0027397addb6","year":2024},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.165846Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:5535975690bf9d143355c8290bfba59929ac250e3e10325b611a810729a8d7ce","observation_id":"902a1e79-b8af-4af6-a343-ea9534ce3109","resolution":{"observed_at":"2026-08-06T19:23:15.594901Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.577209Z","title":"Visually-prompted language model for fine-grained scene graph generation in an open world","venue":null,"work_id":"68232850-b5c8-4e1d-b340-d600ad394d08","year":2023},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.169859Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:57c90eddc058e3d4cef419a493a825e8d7e6e029e8eea6d9a0ffc6b5ae3d8d92","observation_id":"8e4df6da-59e2-4c70-822f-4c0ff33ff213","resolution":{"observed_at":"2026-08-06T19:23:15.581445Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.563163Z","title":"Zero-shot scene graph generation with knowledge graph completion","venue":null,"work_id":"0f1406f1-5968-4438-8e18-db728ffc9468","year":2022},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.174133Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:56e5243252a33bd089aff391df1a90bf8655bdd68d54c1ba8fd093474105a2c6","observation_id":"5d939c45-932a-4792-8759-bff0009cda44","resolution":{"observed_at":"2026-08-06T19:23:15.567311Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.548719Z","title":"Graph transformer networks.Advances in neural information processing systems, 32, 2019","venue":null,"work_id":"7f7f3b0f-98c4-4211-b833-31f925168cd6","year":2019},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.178265Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:c207dfc5cacf5e6438deb22450e5bb9ea920f4695cea2389289549458cf767af","observation_id":"8e7f760f-991e-459f-a5ae-6cb1bcb1b281","resolution":{"observed_at":"2026-08-06T19:23:15.553469Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.532192Z","title":"Open-vocabulary object detection using captions","venue":null,"work_id":"5ca2cea2-7b3f-4cd6-9d15-3e7a618a80ce","year":2021},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.182074Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:43591abeab287f1ca1b93f5bd70de3cf9ad86c9ea0875d8ff87b1fb8dbc0cbe8","observation_id":"b8c6cdae-8449-497f-8302-9afb75958538","resolution":{"observed_at":"2026-08-06T19:23:15.538385Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.517301Z","title":"Neural motifs: Scene graph parsing with global con- text","venue":null,"work_id":"b75aa55f-074b-4398-8a3b-602ee640764d","year":2018},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.185933Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:340f40e99a1a4452387a9e359ea5b31f54e4a4c0b293da23da1528e2e535209e","observation_id":"7bd12bce-0612-4ee6-b087-e9c50ff3e267","resolution":{"observed_at":"2026-08-06T19:23:15.521912Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.501400Z","title":"gddim: Generalized denoising diffusion implicit models","venue":null,"work_id":"37a53416-b816-4527-9120-352dbbda12d4","year":2023},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.189859Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:8b80696627ee3f9f8062a96c00da7d85700c0bda4343220fc4fe0a6e67dee8ec","observation_id":"f2a338ee-7e70-421c-a661-584e46f95458","resolution":{"observed_at":"2026-08-06T19:23:15.505750Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.484650Z","title":"Learning to generate language- supervised and open-vocabulary scene graph using pre- trained visual-semantic space","venue":null,"work_id":"ae0edc6f-4085-4068-b087-ea46bf1f7bed","year":null},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.193911Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:2399dc0566711826ff2c0f628a811d704eebef74ea44eeb18a45d5e18dfc17b6","observation_id":"be0ccd3e-d26a-45eb-84df-4c2e1e4a53dd","resolution":{"observed_at":"2026-08-06T19:23:15.489519Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.470181Z","title":"Unleashing text-to-image diffusion models for visual perception","venue":null,"work_id":"5d921de1-be2d-4711-a406-278569707437","year":null},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.198533Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:945ffa1ebe87116863a1422323d97142ec2ad5977221491fd8ebf7763b85044c","observation_id":"5b940a59-c365-4b65-bdc2-ffacc8599fe2","resolution":{"observed_at":"2026-08-06T19:23:15.474522Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.456081Z","title":"Prototype-based embedding network for scene graph generation","venue":null,"work_id":"5b979233-ffcd-4438-954e-3fb3af20d335","year":2023},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.203065Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:a23edfff2f332bb368f2453d281bf2afae78b94d9731be4f8b80e37cdb4937e7","observation_id":"51a6b8dd-58f2-443a-a97b-7e8724537c58","resolution":{"observed_at":"2026-08-06T19:23:15.460294Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.441736Z","title":"Hilo: Ex- ploiting high low frequency relations for unbiased panoptic scene graph generation","venue":null,"work_id":"a614e736-b5e3-4cef-a8ad-eef341e62e9c","year":2023},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.207023Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:fe8227a63ce212bcf61e898ac2f963bb618b367d1845291a16708179b23bf060","observation_id":"5879cd9c-68e7-451a-b7cb-4855e7770920","resolution":{"observed_at":"2026-08-06T19:23:15.446010Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.425312Z","title":"Openpsg: Open-set panoptic scene graph generation via large multimodal models","venue":null,"work_id":"acd07e0c-3f75-4e13-aea3-73cc36a5595f","year":2024},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.210670Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:faa934607a32a98779baf53939539ade61c70aecc5226be61a50e3367eb84344","observation_id":"c075dfd0-34e4-4bf1-99c0-0b6cf64573f2","resolution":{"observed_at":"2026-08-06T19:23:15.431443Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-08T23:01:31.374352Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning"},"reference_resolution":{"displayed":68,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":7,"verified_exact":0,"verified_fuzzy":61},"total_outbound_references":68},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 68 of 68 outbound references and 0 inbound Pith citation observations for arXiv:2507.05798."}