{"as_of":"2026-08-09T10:29:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:c08ebd726385132568eda6e93fbbbff6f4b5c16a9f99b5a0910df7615b2cc581","coverage":[{"denominator":72,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":72,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T14:06:09.067431Z","state":"measured"},{"denominator":74,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":74,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-27T17:30:48.848277Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-03T00:07:27.999553Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"cited_work":{"arxiv_id":"2505.20106","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.20106","snapshot_observed_at":"2026-07-03T00:07:27.999553Z","title":"OvSGTR: Fully Open-V ocabulary Scene Graph Generation","venue":null,"work_id":"1823271a-6f7e-4b6f-9b06-0dbcf23cf218","year":2025},"citing_paper":{"arxiv_id":"2605.16932","last_updated":"2026-05-16T10:59:51Z","snapshot_observed_at":"2026-08-02T05:26:27.385556Z","submitted_at":"2026-05-16T10:59:51Z","title":"MORN: Metacognitive Object-Goal Regulation for Resource-Rational Long-Horizon Navigation","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-19T20:26:59.407002Z"},"links":{"cited_paper":"/paper/2505.20106","citing_paper":"/paper/2605.16932"},"observation_digest":"sha256:09dfc36076e08212e18ae1cd9f373385a4d4757524ead370f47715f554b75e52","observation_id":"6c41ff57-1027-4708-b114-5749d44553c7","resolution":{"observed_at":"2026-05-19T20:27:49.619298Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"cited_work":{"arxiv_id":"2505.20106","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.20106","snapshot_observed_at":"2026-07-03T00:07:27.999553Z","title":"OvSGTR: Fully Open-V ocabulary Scene Graph Generation","venue":null,"work_id":"1823271a-6f7e-4b6f-9b06-0dbcf23cf218","year":2025},"citing_paper":{"arxiv_id":"2606.09368","last_updated":"2026-08-06T11:38:59Z","snapshot_observed_at":"2026-08-09T10:12:16.011962Z","submitted_at":"2026-06-08T11:40:48Z","title":"PhysScene: A Scene Graph Dataset for Scientific Visual Reasoning in Physics Experiments","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-27T17:30:48.848277Z"},"links":{"cited_paper":"/paper/2505.20106","citing_paper":"/paper/2606.09368"},"observation_digest":"sha256:5e06e4e580e75ea0d655cc3a2ba423b4f0f8490d5e3bee9fd318fd3102bdd77d","observation_id":"a4efc673-b638-4916-9f49-21b1e35a1f6d","resolution":{"observed_at":"2026-07-03T00:07:28.000970Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2505.20106/citation-record","integrity":"/paper/2505.20106/integrity","json":"/paper/2505.20106/citation-record.json","paper":"/paper/2505.20106"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:19.445728Z","title":"Scene graph generation by iterative message passing,","venue":null,"work_id":"4e2429ec-ada6-41a7-8bd1-da6eb0b97ad2","year":2017},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:01.386709Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:7b120daf6375efafa8abfd16d7038436fa1b86d05b8e9960d9fa8908f39900b2","observation_id":"cdee7b1c-498c-4ce6-9dd2-5ae7857f44d3","resolution":{"observed_at":"2026-08-07T14:06:19.512478Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:19.283603Z","title":"Neural motifs: Scene graph parsing with global context,","venue":null,"work_id":"b822ca7c-5e70-4e20-a328-36db830aeb89","year":2018},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:01.463835Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:0832f1bf7b98f284fe2a72d1a989dd6970eb4669c7b110e65e35043335a90843","observation_id":"bb9489d9-b38b-4f4e-8e75-e4d7fc3213c0","resolution":{"observed_at":"2026-08-07T14:06:19.353704Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:19.121077Z","title":"Learning to compose dynamic tree structures for visual contexts,","venue":null,"work_id":"44845ebc-41e9-41e4-a1a4-6d1d9f5579dd","year":2019},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:01.611625Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:9ff2b5d841fdc700f3f7786021776f5b2c3751b2ae6144c4a0b60ccb2a9a77f6","observation_id":"2d989bd4-5f20-45b0-a41d-9f653430d3fa","resolution":{"observed_at":"2026-08-07T14:06:19.190802Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:18.897449Z","title":"Unbiased scene graph generation from biased training,","venue":null,"work_id":"f266d227-9edc-4db9-bb71-b40403cec315","year":2020},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:01.746840Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:52b5d87aadb5938ac258ac84610064af5a4d53dc9ed30c437fd64ff803b12383","observation_id":"7d239a10-0c6d-433e-ad0c-e599fbc3e9ec","resolution":{"observed_at":"2026-08-07T14:06:18.992266Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:18.731691Z","title":"Recovering the unbiased scene graphs from the biased ones,","venue":null,"work_id":"e56f7e1f-5b3a-4eef-b1e4-66e55820c4af","year":2021},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:01.878744Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:9567b184910202ded4059dffce2285d9bf894c5a9bbe160067aadff2119ea3ef","observation_id":"698bbbe9-b9fb-4ded-8bb2-8a8601772183","resolution":{"observed_at":"2026-08-07T14:06:18.801403Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:18.514824Z","title":"Bipartite graph network with adaptive message passing for unbiased scene graph generation,","venue":null,"work_id":"34aa4fbe-d775-4be3-bc29-4add5fd8f027","year":2021},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:02.011294Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:baf2ef0e92260abb611e999dcf674302726be6425bec7c084a31f78b9544e3a0","observation_id":"e858478d-b1ce-41e6-8749-f696b0d757bb","resolution":{"observed_at":"2026-08-07T14:06:18.602538Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:18.153717Z","title":"Graphical contrastive losses for scene graph parsing,","venue":null,"work_id":"cebbff48-0cdb-4745-89f8-94a306b7a83e","year":2019},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:02.243722Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:4cde07736274b656e60807b7ebe52650ee71b14a4993e50199a83e0cb85ef1ee","observation_id":"cf3d4a51-d43c-49d7-8d61-f39be934a8fb","resolution":{"observed_at":"2026-08-07T14:06:18.248992Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:18.040121Z","title":"Towards open-vocabulary scene graph generation with prompt-based finetuning,","venue":null,"work_id":"03fbae29-e3aa-4d7b-9c67-a383a32455a0","year":2022},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:02.335243Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:fcb0fd5ec46ae206d5d738a77d8ebf94b6f5d7d3b8057e20a57f47c57c10d486","observation_id":"ba4e570b-ca0d-435a-960b-7f75fc2a087b","resolution":{"observed_at":"2026-08-07T14:06:18.081171Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:17.863976Z","title":"Learning to generate language-supervised and open-vocabulary scene graph using pre-trained visual-semantic space,","venue":null,"work_id":"3d238bda-8966-4b49-825a-2957f7eb78bf","year":2023},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:02.447077Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:5921b0890e5866032cb9192ef12e33d87b52c91ecfa0642620cd3bf41e06673d","observation_id":"f07808aa-fddb-448c-8d7f-0014dec725bb","resolution":{"observed_at":"2026-08-07T14:06:17.945360Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:17.694534Z","title":"Auto-encoding scene graphs for image captioning,","venue":null,"work_id":"0d4d9c5e-0eba-4df3-a3ec-c4ebec8728c9","year":2019},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:02.536362Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:997ce6dc3c8d58c0a5dd8ac06bc08fd7b55bfb487fcf3b6985770ed262b30fe0","observation_id":"102d9862-29e8-4f15-a2a4-8264c7c50420","resolution":{"observed_at":"2026-08-07T14:06:17.772322Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:17.512832Z","title":"Say as you wish: Fine-grained control of image caption generation with abstract scene graphs,","venue":null,"work_id":"e98d545d-a2c1-4b70-ac01-416045213b5f","year":2020},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:02.622653Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:b243cfe6f9ea0ebb057be6d3a09fee27c4fc575fdc25a39138c23dc03976fa27","observation_id":"c6b0d64b-7ea3-4113-a605-cfd7f34e8faf","resolution":{"observed_at":"2026-08-07T14:06:17.595794Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:17.332076Z","title":"Unpaired image captioning via scene graph alignments,","venue":null,"work_id":"bc2c087a-a2ce-4ca4-9c81-4962b734255f","year":2019},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:02.744274Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:c56cd12b3fbad47e373657cf12825d434b6d25db97d2682458f02ca38c4e391d","observation_id":"6a7ba961-975e-4d05-8448-e7a2c2b27c9b","resolution":{"observed_at":"2026-08-07T14:06:17.418253Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:17.158998Z","title":"On the role of scene graphs in image captioning,","venue":null,"work_id":"29dcf21d-fa8d-4962-8102-4a4f179244a0","year":2019},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:02.813653Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:e456d820469ba49823e40ba798c1eaeb35416fe4dd7754c585d9a56c3dbbc730","observation_id":"46c4d1ae-7e1b-40dd-b271-1061f3b29c61","resolution":{"observed_at":"2026-08-07T14:06:17.242454Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:16.989727Z","title":"In defense of scene graphs for image captioning,","venue":null,"work_id":"be708bac-9928-465e-9527-cd455343635e","year":2021},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:02.891163Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:896676fe0e430b951934de73f4f578cec5964e7ab1c89e29121a21e49c70efd5","observation_id":"d6126e6f-d065-4feb-b5c5-1032d596110b","resolution":{"observed_at":"2026-08-07T14:06:17.067575Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:16.855215Z","title":"Graph-structured representa- tions for visual question answering,","venue":null,"work_id":"e4431dc8-61a7-4778-927d-59b145321fef","year":2017},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:03.013076Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:99644e3d4567ef4a5c427f5cd86ebc45b884e9564443f56ddda1da6d1e18fd74","observation_id":"712ef79c-afe3-406b-8a35-bbf77aa6428a","resolution":{"observed_at":"2026-08-07T14:06:16.913736Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:16.712555Z","title":"Lightweight visual question answering using scene graphs,","venue":null,"work_id":"c7323c25-eab6-43c3-a4d1-9d7d85d875c4","year":2021},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:03.149625Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:ffb1c832644f2a7c48e995b4afff9836080a05b92da8c96345d6a361a8591e26","observation_id":"f8bd5f37-424c-4cda-bc51-95710d36ebea","resolution":{"observed_at":"2026-08-07T14:06:16.784108Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:16.576676Z","title":"Robotvqa - A scene-graph- and deep-learning-based visual question answering system for robot manipulation,","venue":null,"work_id":"ee967200-cddf-40fd-a599-327154e79439","year":2020},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:03.247905Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:7849e13d7d80cf9ff51ddd940df095e90b9a7fea9ece10588d7aad7fd6c9ff59","observation_id":"0c0e1d6e-1ca4-412e-a57f-85b5bfa87696","resolution":{"observed_at":"2026-08-07T14:06:16.640986Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:16.333232Z","title":"Visual question answering over scene graph,","venue":null,"work_id":"baeec614-1c33-4314-b624-dc57545f0278","year":2019},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:03.343070Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:b842a392cb2427d55261fa48d4692a8dde92e09e68b8572783a1774ba3f19a15","observation_id":"6861e06a-199b-471a-b2b2-f093bb79c389","resolution":{"observed_at":"2026-08-07T14:06:16.492477Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:16.002271Z","title":"Image generation from scene graphs,","venue":null,"work_id":"bd74e517-0e4f-403e-9492-ec2ce00c50f3","year":2018},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:03.474269Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:ae572feb8b31c3a9e711d6f4203d44bfd6ed5d2bcdaab3b6aaefc5f599b1a49e","observation_id":"910f75e1-89e1-4fb6-a5b8-9f64030b3953","resolution":{"observed_at":"2026-08-07T14:06:16.178687Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2211.11138","last_updated":"2022-11-21T01:11:19Z","snapshot_observed_at":"2026-08-01T18:48:31.278326Z","submitted_at":"2022-11-21T01:11:19Z","title":"Diffusion-Based Scene Graph to Image Generation with Masked Contrastive Pre-Training","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2211.11138","snapshot_observed_at":"2026-08-07T14:06:03.652422Z","title":"Diffusion-based scene graph to image gener- ation with masked contrastive pre-training,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:03.652422Z"},"links":{"cited_paper":"/paper/2211.11138","citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:3454ebea4073b13baa5fe09ea067861e2d2d136cae2404738f5938fae15e38d6","observation_id":"44f50c3c-ef0f-4bad-a5ed-2cbb13c45c3b","resolution":{"observed_at":"2026-08-07T14:06:03.652422Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.17846","last_updated":"2024-06-03T17:12:25Z","snapshot_observed_at":"2026-08-03T07:21:49.675852Z","submitted_at":"2024-03-26T16:36:43Z","title":"Hierarchical Open-Vocabulary 3D Scene Graphs for Language-Grounded Robot Navigation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.17846","snapshot_observed_at":"2026-08-07T14:06:03.814090Z","title":"Hier- archical open-vocabulary 3d scene graphs for language-grounded robot navigation,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:03.814090Z"},"links":{"cited_paper":"/paper/2403.17846","citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:c574fa53bbc144e24c4229a367c1c01598df13133051f2e1f26d2295e9c602dd","observation_id":"a7cb4a77-d9b7-4fb0-9725-bae17d463b84","resolution":{"observed_at":"2026-08-07T14:06:03.814090Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.08189","last_updated":"2024-10-10T17:57:19Z","snapshot_observed_at":"2026-08-07T23:58:34.214288Z","submitted_at":"2024-10-10T17:57:19Z","title":"SG-Nav: Online 3D Scene Graph Prompting for LLM-based Zero-shot Object Navigation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.08189","snapshot_observed_at":"2026-08-07T14:06:03.978487Z","title":"Sg-nav: Online 3d scene graph prompting for llm-based zero-shot object navigation,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:03.978487Z"},"links":{"cited_paper":"/paper/2410.08189","citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:fee76400f1506ba1c0de73a3353932087ec441048f21c776698c610768891cf4","observation_id":"7d4e668a-00db-4e27-869e-1692add75a64","resolution":{"observed_at":"2026-08-07T14:06:03.978487Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:15.731798Z","title":"Scenegraphloc: Cross-modal coarse visual localization on 3d scene graphs,","venue":null,"work_id":"f8625d8a-bf62-438f-95e4-5e6055e6b180","year":2024},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:04.132246Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:3944535917d640b5c4e52eca73a4c5eb228662abf8afdc9eaa44e63d16423d1a","observation_id":"bcceb78d-83be-4c1e-b2f9-841f4a6d43fc","resolution":{"observed_at":"2026-08-07T14:06:15.875772Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:15.449723Z","title":"Learning to generate scene graph from natural language supervision,","venue":null,"work_id":"42ce8119-952d-4b6e-9a97-67a30d30a27d","year":2021},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:04.264776Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:505e9edf50098b7c61d30acd19a457c4ef68bb684955b0156e0b70348e6db9ca","observation_id":"719b043d-76b2-40c5-9fca-a47039705d96","resolution":{"observed_at":"2026-08-07T14:06:15.568533Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:15.218587Z","title":"Integrating object-aware and interaction-aware knowledge for weakly supervised scene graph generation,","venue":null,"work_id":"095e0ed3-7c3b-4ad6-b371-b37d36e54542","year":2022},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:04.360178Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:9007c455e12a73fed211e294428f68a789c442b3e66542d7e2aa4850635fc19b","observation_id":"13574a47-022e-49f5-9acd-662eaf112166","resolution":{"observed_at":"2026-08-07T14:06:15.328480Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04314","last_updated":"2024-06-02T11:32:19Z","snapshot_observed_at":"2026-07-06T16:58:19.652601Z","submitted_at":"2023-12-07T14:11:00Z","title":"GPT4SGG: Synthesizing Scene Graphs from Holistic and Region-specific Narratives","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04314","snapshot_observed_at":"2026-08-07T14:06:04.444777Z","title":"GPT4SGG: Synthe- sizing scene graphs from holistic and region-specific narratives,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:04.444777Z"},"links":{"cited_paper":"/paper/2312.04314","citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:52125d5827af9afd21b7779893ba00db6e15de75c8f8662fcd2f35a608d01c93","observation_id":"8f3bcb98-2e3b-4011-a13b-2d42917502c0","resolution":{"observed_at":"2026-08-07T14:06:04.444777Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:15.062820Z","title":"Open-vocabulary object detection using captions,","venue":null,"work_id":"6a364988-f65b-4481-92d3-943202bd272b","year":2021},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:04.521096Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:e38969ba77ba363b8f94e78091e0d1fb886f49b1cd6187775e4d45e3ab6b7778","observation_id":"75f732c1-65c1-4422-93cc-875d30e8a8cb","resolution":{"observed_at":"2026-08-07T14:06:15.118860Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:14.906832Z","title":"Aligning bag of regions for open-vocabulary object detection,","venue":null,"work_id":"fd83753a-e22f-43fb-8cfa-56bfe3c57498","year":2023},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:04.611111Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:69a86c2fcd04aed3b8fd1418dee44bb6bafdc4364100cc556d784c87065f47cf","observation_id":"6daa01e2-ff05-4180-a8c1-a59046716e65","resolution":{"observed_at":"2026-08-07T14:06:14.978269Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:14.791064Z","title":"Grounded language-image pre-training,","venue":null,"work_id":"f23c2754-7f3e-4cb9-8c0e-7f6875451087","year":2022},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:04.713894Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:ced313293c44b1c37d61d761f67e277b2beaca21503d37f4a38d38bf46bebf9a","observation_id":"f4b526ed-a1fb-4ca8-8971-ff943c67c27d","resolution":{"observed_at":"2026-08-07T14:06:14.842015Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:14.640364Z","title":"Regionclip: Region-based language- image pretraining,","venue":null,"work_id":"14e6bcae-53db-47a1-ac1e-739bd2c82e10","year":2022},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:04.778671Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:8f944fd707c8ed596de5aefb09e7a71828b8031c69abdb7e44a5917c084a33df","observation_id":"7c26f606-969e-48a4-9a84-85ae3be47776","resolution":{"observed_at":"2026-08-07T14:06:14.708747Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:14.497332Z","title":"Learning to prompt for open-vocabulary object detection with vision-language model,","venue":null,"work_id":"267b268b-00d5-4050-a6cd-5701af308111","year":2022},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:04.865742Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:2a7bb2cc51a76995c90d62c7b50c4368062c4621cb6f020050ca945fd53ebfbe","observation_id":"e5a12c88-11f3-40be-bd70-089f3c99d99d","resolution":{"observed_at":"2026-08-07T14:06:14.563025Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:14.353898Z","title":"Scene graph parser,","venue":null,"work_id":"5b94d7e8-0a47-4eae-a09d-9f33049e45c5","year":2022},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:04.975100Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:652991ea789bb545010c0a75706a85b08f9a2c2e6ada087d5de718c95b3e8d63","observation_id":"5f143c4c-2a2f-447e-8384-30f7816d52a5","resolution":{"observed_at":"2026-08-07T14:06:14.419755Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:14.232487Z","title":"Scene graph generation from objects, phrases and region captions,","venue":null,"work_id":"0411792c-5e8c-462b-a587-926c5dc1f11a","year":2017},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:05.098519Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:57acf0237d07d11dd8a48e94c4382e0f0b106924aed81924a7cce46da9835d11","observation_id":"68bca05c-7128-4e54-a4e4-cf1a1aeb4dc5","resolution":{"observed_at":"2026-08-07T14:06:14.276576Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:18.336309Z","title":"Knowledge-embedded routing network for scene graph generation,","venue":null,"work_id":"0d4daaf4-362e-483a-9ef3-3e695103b5bf","year":2019},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:05.176216Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:2b563607f81610c59ae2b5f4f3bb214d7b1a44b528c265674ee8a200bdcf0099","observation_id":"7c904e98-17d2-45c8-9fa0-158f3aac65a6","resolution":{"observed_at":"2026-08-07T14:06:18.419833Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:14.083841Z","title":"Sgtr: End-to-end scene graph generation with transformer,","venue":null,"work_id":"a24a704e-2a37-4092-8f12-b03459db2dc6","year":2022},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:05.302846Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:7138b120c1367604450f90bc0e26ff8f1ea99ac793bd54a605c4672d2b1b7a60","observation_id":"d04a9a2d-7d63-405e-8f4c-2f942e33f315","resolution":{"observed_at":"2026-08-07T14:06:14.156470Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:13.979383Z","title":"Iterative scene graph generation,","venue":null,"work_id":"553ac578-8cc9-4e5c-8ae1-b3d399ea8b5a","year":2022},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:05.397991Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:f2967dbf122cda0c2f56a84728e82a24b0113d1971fa56bf17bd3e1843106b6d","observation_id":"fe14bb51-9adb-4ce5-bf79-a1f560062aab","resolution":{"observed_at":"2026-08-07T14:06:14.009172Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:13.837298Z","title":"Reltr: Relation transformer for scene graph generation,","venue":null,"work_id":"b5d2fb25-69e6-4fd7-afe5-2d69df278eb7","year":2023},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:05.483722Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:460e8398c11ec901a1bc6e3c376fe95135cf79fd51f628e690024c37b6150847","observation_id":"d437da9e-d945-45b0-af75-5a85e280fe39","resolution":{"observed_at":"2026-08-07T14:06:13.910622Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:13.715515Z","title":"Unbiased scene graph generation via two-stage causal modeling,","venue":null,"work_id":"ff53b4fb-52e1-462d-8469-8dc6a78e2cc6","year":2023},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:05.570808Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:df2327f8730c0f8a26803591e83572cd1d7c1386e35e6c94d8949d718727e643","observation_id":"e935d187-0f34-43c2-aec2-846b981d37be","resolution":{"observed_at":"2026-08-07T14:06:13.780426Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:13.545521Z","title":"Fast contextual scene graph generation with unbiased context augmentation,","venue":null,"work_id":"fecd0e15-937b-4ed3-b31f-6180087f9682","year":2023},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:05.732333Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:71446b0e523b73410e3fa84237fc89903b76e6e7a0e5fde49cb91399acf9a05c","observation_id":"7e92f33b-87f3-45a3-9090-344ff2f68c8e","resolution":{"observed_at":"2026-08-07T14:06:13.625301Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:13.434412Z","title":"Semantic diversity-aware prototype-based learning for unbiased scene graph generation,","venue":null,"work_id":"63bea7a8-f282-4bf1-9ea7-68fc2cc8bd71","year":2024},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:05.861402Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:e7ffadc532d6944a4b8587560a86773954b1e80e08a1fe94558757fb65080223","observation_id":"c1a8d885-c195-465f-8adf-40ba9bc32073","resolution":{"observed_at":"2026-08-07T14:06:13.494396Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:05.948903Z","title":"Faster r-cnn: Towards real-time object detection with region proposal networks,","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:05.948903Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:3a52ff3f080454a33dc98e795d1ee19d7149ed6b14a8ff15a5c4a2044d140194","observation_id":"4dc38357-9ffa-4291-8bab-ddb47d7f6e35","resolution":{"observed_at":"2026-08-07T14:06:05.948903Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15435","last_updated":"2025-05-26T11:55:56Z","snapshot_observed_at":"2026-07-06T19:55:51.415524Z","submitted_at":"2024-11-23T03:40:25Z","title":"What Makes a Scene ? Scene Graph-based Evaluation and Feedback for Controllable Generation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.15435","snapshot_observed_at":"2026-08-07T14:06:06.096668Z","title":"What makes a scene? scene graph-based evaluation and feedback for controllable generation,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:06.096668Z"},"links":{"cited_paper":"/paper/2411.15435","citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:5216dba9a5146327f3e6d114b0bd9da184defa53ec1ed81b905ff7f0887f6485","observation_id":"9212e703-b1af-4a6b-bfc1-e00ae40ceed4","resolution":{"observed_at":"2026-08-07T14:06:06.096668Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:13.327242Z","title":"Learning transferable visual models from natural language supervi- sion,","venue":null,"work_id":"f36049b2-3bae-4ee4-858a-8251201f62d7","year":2021},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:06.205169Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:c013bf41218ae321aed16cc360b487669e440ec9a365c629fabe0019beb80155","observation_id":"daccf754-f33f-47b7-ba14-4939ac8e0264","resolution":{"observed_at":"2026-08-07T14:06:13.376646Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.05499","last_updated":"2024-07-19T06:00:41Z","snapshot_observed_at":"2026-07-06T15:00:58.804337Z","submitted_at":"2023-03-09T18:52:16Z","title":"Grounding DINO: Marrying DINO with Grounded Pre-Training for Open-Set Object Detection","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.05499","snapshot_observed_at":"2026-08-07T14:06:06.308339Z","title":"Grounding DINO: marrying DINO with grounded pre-training for open-set object detection,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:06.308339Z"},"links":{"cited_paper":"/paper/2303.05499","citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:444d0e243df83bb146cc8edfd89ddfa1be8f6885f2e511a053b64f171951cc76","observation_id":"ee357b2c-8fa4-48b8-a7f6-5ea001e9b4ae","resolution":{"observed_at":"2026-08-07T14:06:06.308339Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1504.00325","last_updated":"2015-04-03T20:21:16Z","snapshot_observed_at":"2026-08-04T18:05:27.145522Z","submitted_at":"2015-04-01T18:13:43Z","title":"Microsoft COCO Captions: Data Collection and Evaluation Server","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1504.00325","snapshot_observed_at":"2026-08-07T14:06:06.382309Z","title":"Microsoft COCO captions: Data collection and evaluation server,","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:06.382309Z"},"links":{"cited_paper":"/paper/1504.00325","citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:c0d3c14e9c1c61779a5c56c167140bb7105704ee71538944989aee3b0aaeacd4","observation_id":"1ef1ab04-0b6d-42cb-a269-e4804826c615","resolution":{"observed_at":"2026-08-07T14:06:06.382309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:13.185644Z","title":"Open-vocabulary object detection via vision and language knowledge distillation,","venue":null,"work_id":"df6184d3-0581-4d21-82f8-2b960f0b4122","year":2022},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:06.495719Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:c1fc483214aea93a2fc47506676e167cb4483438b13bcbc60b340261ee5301b3","observation_id":"df0a26ef-d565-4d4c-96c1-9d6182578379","resolution":{"observed_at":"2026-08-07T14:06:13.251945Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:13.007473Z","title":"Scaling open-vocabulary image segmentation with image-level labels,","venue":null,"work_id":"a2def186-56e0-4adb-8a60-a8b35ee54343","year":2022},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:06.578336Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:96418ba67f2d3ea08dce5ef17ab8ed2bae4c0c06bf10c358aaf3f3dd22a293e5","observation_id":"ca382652-816d-4088-acf0-ae1b707b2be6","resolution":{"observed_at":"2026-08-07T14:06:13.088490Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.08472","last_updated":"2021-09-17T11:21:34Z","snapshot_observed_at":"2026-07-06T11:48:42.000083Z","submitted_at":"2021-09-17T11:21:34Z","title":"ActionCLIP: A New Paradigm for Video Action Recognition","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2109.08472","snapshot_observed_at":"2026-08-07T14:06:06.655155Z","title":"Actionclip: A new paradigm for video action recognition,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:06.655155Z"},"links":{"cited_paper":"/paper/2109.08472","citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:835e0067a7ea34b64788689bb4c8ecc7ce0df7dc10809ffe2a4732f2dd929e48","observation_id":"ca03ad4d-a30e-4554-a856-6b3bd56dea65","resolution":{"observed_at":"2026-08-07T14:06:06.655155Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:12.820760Z","title":"From pixels to graphs: Open-vocabulary scene graph generation with vision-language models,","venue":null,"work_id":"2f7d60de-4113-4040-8ccc-2beb53b9363d","year":2024},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:06.769749Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:acdc0979314944fc5dcca3614347687e4a8870410818255f233b0974c2fadf91","observation_id":"b654f10c-9127-4f0b-b57d-e80deea9cac9","resolution":{"observed_at":"2026-08-07T14:06:12.911201Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:12.656448Z","title":"Towards open vocabulary learning: A survey,","venue":null,"work_id":"001a4b3a-3d90-478b-9e05-156b9241bfcc","year":2024},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:06.867087Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:217ec0549d17e1045baebcc7768e5ab9aa6ebc2a51a5ed4871339b4fd995a895","observation_id":"9e5bd9b7-faed-407c-8ef9-4640feb433a9","resolution":{"observed_at":"2026-08-07T14:06:12.750669Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09220","last_updated":"2024-04-15T02:47:01Z","snapshot_observed_at":"2026-07-06T15:55:21.408850Z","submitted_at":"2023-07-18T12:52:49Z","title":"A Survey on Open-Vocabulary Detection and Segmentation: Past, Present, and Future","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.09220","snapshot_observed_at":"2026-08-07T14:06:06.939775Z","title":"A survey on open-vocabulary detection and segmentation: Past, present, and future,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:06.939775Z"},"links":{"cited_paper":"/paper/2307.09220","citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:980d929f15257859a0968c28581dc13f263b5a784e44556e804fd2d75e5e6ce8","observation_id":"b10b6357-aa9f-44c0-b104-27dd91218382","resolution":{"observed_at":"2026-08-07T14:06:06.939775Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:12.463269Z","title":"LLM4SGG: Large language models for weakly supervised scene graph generation,","venue":null,"work_id":"ece171e8-2f23-4070-a338-cf8f5b4ace08","year":2024},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:07.025428Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:748f87656548ba473d4a4ef4ae5a2161d3869d0236f8833bbc5c9cd70fdc147b","observation_id":"706aa460-dc7a-4c11-b99d-3eef5e3ff097","resolution":{"observed_at":"2026-08-07T14:06:12.565694Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:12.239904Z","title":"GPT-4v(ision) System Card,","venue":null,"work_id":"1f523f2f-1245-41e9-b57f-24feb1e83811","year":2023},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:07.157365Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:46b37156948cde725c29fe605a48ed5c9fc81671a08097ee9dba67528a595aba","observation_id":"cb899a64-2c3e-4981-a4a7-d4f20eb76682","resolution":{"observed_at":"2026-08-07T14:06:12.359205Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:12.013390Z","title":"Expanding scene graph boundaries: fully open-vocabulary scene graph generation via visual-concept alignment and retention,","venue":null,"work_id":"11ad65ec-0f88-4d0c-8c03-3b8efbc9339c","year":2024},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:07.233373Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:4d25d3e4e8f825b99dc1e7ed35f9be7fe29976f2b8345a62afba47cde39ceb1c","observation_id":"001b0fa4-b907-4ff9-9875-38fe4e234bd9","resolution":{"observed_at":"2026-08-07T14:06:12.102556Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:11.828004Z","title":"Swin transformer: Hierarchical vision transformer using shifted windows,","venue":null,"work_id":"09e3e79f-1771-41ab-927d-d349dc9b3108","year":2021},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:07.349389Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:42f478617d71184f87553dd0fa6c75bcf34f32d169095ea3576cd12b48cba8fb","observation_id":"d40878fe-ea6a-4972-806f-235636f6ba4f","resolution":{"observed_at":"2026-08-07T14:06:11.916722Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:11.641295Z","title":"BERT: pre-training of deep bidirectional transformers for language understanding,","venue":null,"work_id":"09408e28-f1e7-47b8-bc64-af6bcc56f607","year":2019},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:07.436553Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:7f778fd83b3ffeeb33aaca42315034c8ca4b3cf86e498315c15fdc2f48da0a4c","observation_id":"48a191a7-e43a-4b4b-8c9a-fc837c1bc116","resolution":{"observed_at":"2026-08-07T14:06:11.731059Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:11.444493Z","title":"Deformable DETR: deformable transformers for end-to-end object detection,","venue":null,"work_id":"2e1d1708-ce92-4049-b497-d54d176a5dd9","year":2021},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:07.557490Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:847abe82d8c938c30d04a918de89fe28d5169de366dcdc6f3b58a8b9def2b903","observation_id":"542926b9-d50a-43a2-859d-3864ecc202d8","resolution":{"observed_at":"2026-08-07T14:06:11.532375Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:11.288201Z","title":"Generalized intersection over union: A metric and a loss for bounding box regression,","venue":null,"work_id":"9affd302-3c09-475a-9264-73e1def496b3","year":2019},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:07.672161Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:1a3b82e18f431c966d4cfe0923522c919b2f33365907f18d0801f32e40e8724b","observation_id":"dc89014e-d49d-4cdb-8723-9b15616533ac","resolution":{"observed_at":"2026-08-07T14:06:11.333322Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:11.112893Z","title":"Focal loss for dense object detection,","venue":null,"work_id":"8877cb04-24fc-4f8c-ad8e-afa094f2ab87","year":2017},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:07.775201Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:19a09516982a51779cb4abcba6a4811f6829acca2dde3e494ddddb022df25d9a","observation_id":"705cdc96-3f30-4e69-becb-f9b99d6bc8ef","resolution":{"observed_at":"2026-08-07T14:06:11.190363Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-07T14:06:07.946971Z","title":"Gpt-4 technical report,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:07.946971Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:68d26af1341bda5fc3f639acb86b01367b51300d4c240f41d1b35099e5ecd2bf","observation_id":"069e383a-f387-4379-8edc-15c9d4d56a77","resolution":{"observed_at":"2026-08-07T14:06:07.946971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.05530","last_updated":"2024-12-16T17:39:39Z","snapshot_observed_at":"2026-07-06T17:41:42.995949Z","submitted_at":"2024-03-08T18:54:20Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.05530","snapshot_observed_at":"2026-08-07T14:06:08.035323Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:08.035323Z"},"links":{"cited_paper":"/paper/2403.05530","citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:dbefe504856f37ce9dfd85021e7686abe4bdd3b7d1ba8e671455f64551727ccc","observation_id":"5d455cc7-4ad7-4b5a-bb31-c054fd9ef4b3","resolution":{"observed_at":"2026-08-07T14:06:08.035323Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:10.904042Z","title":"Gqa: A new dataset for real-world visual reasoning and compositional question answering,","venue":null,"work_id":"dde9bd9a-aa5e-4fdc-bb8c-c3e9208f8006","year":2019},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:08.143995Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:4279a4e90ed8b1cf7d9d83fc1c2ac2816ef980388d373f87b1d7efe5b12c9128","observation_id":"01046d1e-206e-46af-8ca5-23519cf25bc6","resolution":{"observed_at":"2026-08-07T14:06:10.982766Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:10.745699Z","title":"Visual genome: Connecting language and vision using crowdsourced dense image annotations,","venue":null,"work_id":"7ef6b0ab-75fc-4d37-a62a-0178be8d446b","year":2017},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:08.245435Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:87ca626e23a885ca2e272f18a932c22dc717fc4ee31db306f2e8db9c073c678d","observation_id":"1abba453-9b9d-4aaa-81cb-75ed808a26e4","resolution":{"observed_at":"2026-08-07T14:06:10.820631Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:10.557423Z","title":"Stacked hybrid-attention and group collaborative learning for unbiased scene graph generation,","venue":null,"work_id":"48217444-6b81-487e-8188-885f0d504127","year":2022},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:08.332388Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:9d47cb246275902741b1907a1dac4f93231fc6454665ccdf6194d3488bc60bc1","observation_id":"a559b35e-0c62-4143-8680-c82ffe75e257","resolution":{"observed_at":"2026-08-07T14:06:10.625054Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:10.358963Z","title":"Vision relation transformer for unbiased scene graph generation,","venue":null,"work_id":"831f4b61-a8e6-40de-9eec-510d891d1131","year":2023},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:08.435412Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:3c85f280f4cf4c1b66dbc20fe87ba0849727e0117e5e8cb2efab8345fb3be978","observation_id":"aa9ed081-3f84-4f1c-adbd-75939cc1d995","resolution":{"observed_at":"2026-08-07T14:06:10.473040Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:08.532477Z","title":"Decoupled weight decay regularization,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:08.532477Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:4dcade2e0881385e15fff65b83c2e5bbeae8a4b7cbd966d1b4a74ddecab13f2c","observation_id":"51c3d11c-4363-4cc2-b94a-0a18b9469057","resolution":{"observed_at":"2026-08-07T14:06:08.532477Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:10.141868Z","title":"Linguistic structures as weak supervision for visual scene graph generation,","venue":null,"work_id":"a413b8b5-baf0-44f0-9689-13d0652f415c","year":2021},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:08.611359Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:ddd94ba7af8c436bf2e8edcfa1b8d8219948bfdc3323f11d07d86ff04991a0d5","observation_id":"18ad3b4f-4909-4273-97cf-ff2589428070","resolution":{"observed_at":"2026-08-07T14:06:10.232457Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:09.919079Z","title":"UNITER: universal image-text representation learning,","venue":null,"work_id":"e520e076-6b63-4e72-9d51-1ae6d3964be1","year":2020},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:08.685373Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:47b5c0fda21515d9901617902405f80144af403c53ca1dbbafc61ca99fa569f8","observation_id":"09da9128-4e40-436e-ab37-ceeb3caffa33","resolution":{"observed_at":"2026-08-07T14:06:10.006059Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:09.745500Z","title":"Hl-net: Heterophily learning network for scene graph generation,","venue":null,"work_id":"a880aac9-cf6a-4636-8f67-e0a910056dba","year":2022},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:08.795013Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:5febe96bc04e130b5b546c952e18f66b1b4649dbc251b57c84d426c3b490c1d5","observation_id":"86803cdc-bbcb-42d1-a26d-7f036b724630","resolution":{"observed_at":"2026-08-07T14:06:09.840428Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:09.536722Z","title":"Fully convolutional scene graph generation,","venue":null,"work_id":"05953740-4b15-4e83-b52f-7c7dcb546ce4","year":2021},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:08.882042Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:51c0f7a96051d8a50880f3ce237a59d9d75e1a9eeca9e89b12e6dce1f3e307e6","observation_id":"b077492e-ffa8-4623-bdff-4eda17086f8c","resolution":{"observed_at":"2026-08-07T14:06:09.615403Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:09.326642Z","title":"Leveraging predicate and triplet learning for scene graph generation,","venue":null,"work_id":"5a41f6d7-b8fd-4b03-bdb3-72a5b172a0d3","year":2024},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:08.966290Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:66b8dd64cace97a78f44d3d98f2af120335a220d274b1954780a99a1c42ad548","observation_id":"c1546d6c-a2e4-4259-bcaa-9eadf4f946b5","resolution":{"observed_at":"2026-08-07T14:06:09.449308Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:06:09.067431Z","title":"Visualizing data using t-sne","venue":null,"work_id":null,"year":2008},"citing_paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:09.067431Z"},"links":{"citing_paper":"/paper/2505.20106"},"observation_digest":"sha256:72052c675d3ca17be3d98caa30ae43e355f9f72b857b599ecfcac8c8dc63cd01","observation_id":"69142701-1c91-4690-a311-5b20cac60090","resolution":{"observed_at":"2026-08-07T14:06:09.067431Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2505.20106","last_updated":"2025-05-26T15:11:23Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-08T02:52:10.331528Z","submitted_at":"2025-05-26T15:11:23Z","title":"From Data to Modeling: Fully Open-vocabulary Scene Graph Generation"},"reference_resolution":{"displayed":72,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":14,"verified_exact":0,"verified_fuzzy":58},"total_outbound_references":72},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 72 of 72 outbound references and 2 inbound Pith citation observations for arXiv:2505.20106."}