{"as_of":"2026-08-15T19:20:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:355a5585de4c46abf4975a042a19cdc2cc1facea17de8b6d06ba2dc9fad76f3a","coverage":[{"denominator":27,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":27,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T05:11:19.983039Z","state":"measured"},{"denominator":28,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":28,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-15T06:32:42.880941+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-01T06:21:09.996386Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":0,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.08553","last_updated":"2025-06-10T08:21:38Z","snapshot_observed_at":"2026-08-15T04:48:41.491398Z","submitted_at":"2025-06-10T08:21:38Z","title":"From Pixels to Graphs: using Scene and Knowledge Graphs for HD-EPIC VQA Challenge","version":1},"cited_work":{"arxiv_id":"2506.08553","doi":"10.48550/arxiv.2506.08553","metadata_source":"arxiv_reference","pith_arxiv_id":"2506.08553","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CoRRabs/2506.08553(June 2025),https://doi.org/ 10.48550/arXiv.2506.08553","venue":"ArXiv.org","work_id":"17482636-2bab-4148-bfb4-76a22458f310","year":2025},"citing_paper":{"arxiv_id":"2606.31187","last_updated":"2026-06-30T06:16:17Z","snapshot_observed_at":"2026-07-07T00:04:58.486133Z","submitted_at":"2026-06-30T06:16:17Z","title":"Learning to Deny: Action Denial in Multimodal Large Language Models","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-07-01T06:21:09.996386Z"},"links":{"cited_paper":"/paper/2506.08553","citing_paper":"/paper/2606.31187"},"observation_digest":"sha256:15a5928d153ce3954ec4edb69348aabed6f31223b30ebc187212f66d0e5e8a29","observation_id":"2f522f67-b64e-4061-b04d-5487d7ea1536","resolution":{"observed_at":"2026-07-01T06:25:26.952478Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2506.08553/citation-record","integrity":"/paper/2506.08553/integrity","json":"/paper/2506.08553/citation-record.json","paper":"/paper/2506.08553"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:11:20.312227Z","title":"Flamingo: a visual language model for few-shot learning","venue":null,"work_id":"3799b78d-67a9-485d-96aa-bcbf9c968266","year":2022},"citing_paper":{"arxiv_id":"2506.08553","last_updated":"2025-06-10T08:21:38Z","snapshot_observed_at":"2026-08-15T04:48:41.491398Z","submitted_at":"2025-06-10T08:21:38Z","title":"From Pixels to Graphs: using Scene and Knowledge Graphs for HD-EPIC VQA Challenge","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T05:11:19.878734Z"},"links":{"citing_paper":"/paper/2506.08553"},"observation_digest":"sha256:9262b5dc78184dc33130c514b56cee4096f7e0ea3fe8aa571df6b97d32bf54c8","observation_id":"c2f6542a-0d1b-4646-bb97-d9376f273ce1","resolution":{"observed_at":"2026-08-07T05:11:20.316550Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:11:20.300076Z","title":"3d scene graph: A structure for unified semantics, 3d space, and cam- era","venue":null,"work_id":"ee373fb5-5b30-4a0a-9f38-943cd2244d81","year":2019},"citing_paper":{"arxiv_id":"2506.08553","last_updated":"2025-06-10T08:21:38Z","snapshot_observed_at":"2026-08-15T04:48:41.491398Z","submitted_at":"2025-06-10T08:21:38Z","title":"From Pixels to Graphs: using Scene and Knowledge Graphs for HD-EPIC VQA Challenge","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T05:11:19.883010Z"},"links":{"citing_paper":"/paper/2506.08553"},"observation_digest":"sha256:acdffdeddb0892b2bbef5398c8599d915aa30102e91e210acc2aded4ae96a05b","observation_id":"be1adb73-2497-4037-9299-eb3f6afb6417","resolution":{"observed_at":"2026-08-07T05:11:20.304373Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:11:20.288157Z","title":"Lamb Artur d’Avila Garcez","venue":null,"work_id":"4e6d2353-9ed8-49d0-950d-988033fd9a3e","year":2023},"citing_paper":{"arxiv_id":"2506.08553","last_updated":"2025-06-10T08:21:38Z","snapshot_observed_at":"2026-08-15T04:48:41.491398Z","submitted_at":"2025-06-10T08:21:38Z","title":"From Pixels to Graphs: using Scene and Knowledge Graphs for HD-EPIC VQA Challenge","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T05:11:19.887059Z"},"links":{"citing_paper":"/paper/2506.08553"},"observation_digest":"sha256:d908b679792d1ebb54913f298166830442421eb6c8657562242708d5f4d43161","observation_id":"4c63f293-f3c6-48e6-bcf6-0f1373285c4c","resolution":{"observed_at":"2026-08-07T05:11:20.291760Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:11:20.276300Z","title":"Comet: Com- monsense transformers for automatic knowledge graph con- struction","venue":null,"work_id":"98b2ebf7-ced9-4d20-99c2-a1ae1077ff52","year":2019},"citing_paper":{"arxiv_id":"2506.08553","last_updated":"2025-06-10T08:21:38Z","snapshot_observed_at":"2026-08-15T04:48:41.491398Z","submitted_at":"2025-06-10T08:21:38Z","title":"From Pixels to Graphs: using Scene and Knowledge Graphs for HD-EPIC VQA Challenge","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T05:11:19.891088Z"},"links":{"citing_paper":"/paper/2506.08553"},"observation_digest":"sha256:0d5fcd4bddfba08133587335b84433dab6881fb27dc0b29d10c2a384ce16d52e","observation_id":"235cf7a3-ff0b-42d6-9078-43480f14a488","resolution":{"observed_at":"2026-08-07T05:11:20.280349Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:11:20.264671Z","title":"Towards Neuro- Symbolic Video Understanding, page 220–236","venue":null,"work_id":"149f86e1-db8f-4355-b764-fdff44af913e","year":2024},"citing_paper":{"arxiv_id":"2506.08553","last_updated":"2025-06-10T08:21:38Z","snapshot_observed_at":"2026-08-15T04:48:41.491398Z","submitted_at":"2025-06-10T08:21:38Z","title":"From Pixels to Graphs: using Scene and Knowledge Graphs for HD-EPIC VQA Challenge","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T05:11:19.895044Z"},"links":{"citing_paper":"/paper/2506.08553"},"observation_digest":"sha256:49084b3b1ef2b20df46c3fc76b522f0ecf5846a3562ade342bbe687fe6c945c4","observation_id":"21240ddd-d8de-4b24-b25c-01d998c275d9","resolution":{"observed_at":"2026-08-07T05:11:20.268509Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:11:20.253044Z","title":"The epic-kitchens dataset: Collection, challenges and base- lines","venue":null,"work_id":"79b7a310-89ab-41d7-a83d-e3935e4da6e8","year":2021},"citing_paper":{"arxiv_id":"2506.08553","last_updated":"2025-06-10T08:21:38Z","snapshot_observed_at":"2026-08-15T04:48:41.491398Z","submitted_at":"2025-06-10T08:21:38Z","title":"From Pixels to Graphs: using Scene and Knowledge Graphs for HD-EPIC VQA Challenge","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T05:11:19.899376Z"},"links":{"citing_paper":"/paper/2506.08553"},"observation_digest":"sha256:3bd3d46ad26dd4700ba920b7927ca24b60f3b04da8915b16ee66a01b7b2d329f","observation_id":"73ff61ec-5855-43d9-b6b6-04b1491bb20a","resolution":{"observed_at":"2026-08-07T05:11:20.256974Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.17580","last_updated":"2023-12-03T18:17:21Z","snapshot_observed_at":"2026-08-12T12:50:51.005571Z","submitted_at":"2023-03-30T17:48:28Z","title":"HuggingGPT: Solving AI Tasks with ChatGPT and its Friends in Hugging Face","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.17580","snapshot_observed_at":"2026-08-07T05:11:19.903965Z","title":"Gemini: Integrating large language models and large vision models for general-purpose multimodal ai","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.08553","last_updated":"2025-06-10T08:21:38Z","snapshot_observed_at":"2026-08-15T04:48:41.491398Z","submitted_at":"2025-06-10T08:21:38Z","title":"From Pixels to Graphs: using Scene and Knowledge Graphs for HD-EPIC VQA Challenge","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T05:11:19.903965Z"},"links":{"cited_paper":"/paper/2303.17580","citing_paper":"/paper/2506.08553"},"observation_digest":"sha256:bf78802f5998f44551ef71a5a10b010c211163e3facb5f64f5afd32696b8643b","observation_id":"28a4f84c-5e7d-4872-9065-b822f31ad9fe","resolution":{"observed_at":"2026-08-07T05:11:19.903965Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:11:20.241365Z","title":"Gemini 2.0 flash","venue":null,"work_id":"871863ac-6e90-4531-b40e-4e947706048d","year":null},"citing_paper":{"arxiv_id":"2506.08553","last_updated":"2025-06-10T08:21:38Z","snapshot_observed_at":"2026-08-15T04:48:41.491398Z","submitted_at":"2025-06-10T08:21:38Z","title":"From Pixels to Graphs: using Scene and Knowledge Graphs for HD-EPIC VQA Challenge","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T05:11:19.908376Z"},"links":{"citing_paper":"/paper/2506.08553"},"observation_digest":"sha256:7e0057d4af77f52e1c75b37042e654e699b126f42ebf1329c467fa4314cf7de6","observation_id":"fa42f414-e5fd-44bc-8d4f-b057a07f5006","resolution":{"observed_at":"2026-08-07T05:11:20.245182Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:11:19.912469Z","title":"Ego4d: Around the world in 3,000 hours of egocentric video","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.08553","last_updated":"2025-06-10T08:21:38Z","snapshot_observed_at":"2026-08-15T04:48:41.491398Z","submitted_at":"2025-06-10T08:21:38Z","title":"From Pixels to Graphs: using Scene and Knowledge Graphs for HD-EPIC VQA Challenge","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T05:11:19.912469Z"},"links":{"citing_paper":"/paper/2506.08553"},"observation_digest":"sha256:9c16e5537729d50720a2e55b4953fadd78ca9ea4edf2628aaa078e1aca1707b7","observation_id":"d8803587-dafd-4457-9d08-c24f064298b6","resolution":{"observed_at":"2026-08-07T05:11:19.912469Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:11:20.223291Z","title":"Learning by asking questions","venue":null,"work_id":"1f6cf3e2-f382-4909-85b0-68e1ad3a9053","year":2019},"citing_paper":{"arxiv_id":"2506.08553","last_updated":"2025-06-10T08:21:38Z","snapshot_observed_at":"2026-08-15T04:48:41.491398Z","submitted_at":"2025-06-10T08:21:38Z","title":"From Pixels to Graphs: using Scene and Knowledge Graphs for HD-EPIC VQA Challenge","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T05:11:19.916164Z"},"links":{"citing_paper":"/paper/2506.08553"},"observation_digest":"sha256:b59df867e3d0a51662ece126b93ed9f4aa5e2d2c7ffa9e8b5e366f153ebcbee6","observation_id":"bdd20501-19e8-4cda-9108-a5bf667b1c25","resolution":{"observed_at":"2026-08-07T05:11:20.226951Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:11:20.212249Z","title":"Patching open- vocabulary vision models with commonsense","venue":null,"work_id":"a42b4239-8fb2-49aa-84da-9c1df0122bc3","year":null},"citing_paper":{"arxiv_id":"2506.08553","last_updated":"2025-06-10T08:21:38Z","snapshot_observed_at":"2026-08-15T04:48:41.491398Z","submitted_at":"2025-06-10T08:21:38Z","title":"From Pixels to Graphs: using Scene and Knowledge Graphs for HD-EPIC VQA Challenge","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T05:11:19.920387Z"},"links":{"citing_paper":"/paper/2506.08553"},"observation_digest":"sha256:dd63338dddec4e1da49864fd1951c3d0550ab5f49d505ac68e7a912d88dac036","observation_id":"81b095c9-5195-41f7-a11d-31a8aefc8eab","resolution":{"observed_at":"2026-08-07T05:11:20.216102Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:11:20.201655Z","title":"Shamma, Michael Bernstein, and Li Fei-Fei","venue":null,"work_id":"998fefb1-9016-45e4-b7e5-7af816632c04","year":2015},"citing_paper":{"arxiv_id":"2506.08553","last_updated":"2025-06-10T08:21:38Z","snapshot_observed_at":"2026-08-15T04:48:41.491398Z","submitted_at":"2025-06-10T08:21:38Z","title":"From Pixels to Graphs: using Scene and Knowledge Graphs for HD-EPIC VQA Challenge","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T05:11:19.924190Z"},"links":{"citing_paper":"/paper/2506.08553"},"observation_digest":"sha256:1eb0fc76b752b25192042e6251e02cec18623acf00886d6c081f6195a4d1753b","observation_id":"9465e5cb-9df3-4a3a-9914-43f908f42393","resolution":{"observed_at":"2026-08-07T05:11:20.205152Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:11:20.190658Z","title":null,"venue":null,"work_id":"c44b4a8f-4379-47a9-beea-3e6f43460d38","year":2024},"citing_paper":{"arxiv_id":"2506.08553","last_updated":"2025-06-10T08:21:38Z","snapshot_observed_at":"2026-08-15T04:48:41.491398Z","submitted_at":"2025-06-10T08:21:38Z","title":"From Pixels to Graphs: using Scene and Knowledge Graphs for HD-EPIC VQA Challenge","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T05:11:19.927911Z"},"links":{"citing_paper":"/paper/2506.08553"},"observation_digest":"sha256:8aeef934643c08b8a639d29ad2a1f975277edef10a2d90cc62a3f646bb118ce4","observation_id":"53b74dbc-71cb-4134-8ecc-e7a59a79144a","resolution":{"observed_at":"2026-08-07T05:11:20.194397Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2201.12086","last_updated":"2022-02-15T05:43:32Z","snapshot_observed_at":"2026-08-09T08:39:37.884261Z","submitted_at":"2022-01-28T12:49:48Z","title":"BLIP: Bootstrapping Language-Image Pre-training for Unified Vision-Language Understanding and Generation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2201.12086","snapshot_observed_at":"2026-08-07T05:11:19.932135Z","title":"Blip: Bootstrapping language-image pre-training for uni- fied vision-language understanding and generation","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.08553","last_updated":"2025-06-10T08:21:38Z","snapshot_observed_at":"2026-08-15T04:48:41.491398Z","submitted_at":"2025-06-10T08:21:38Z","title":"From Pixels to Graphs: using Scene and Knowledge Graphs for HD-EPIC VQA Challenge","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T05:11:19.932135Z"},"links":{"cited_paper":"/paper/2201.12086","citing_paper":"/paper/2506.08553"},"observation_digest":"sha256:d8a17e22377d5230b087ce176916d761047fe8e039c678c4b9cf03efd904ab05","observation_id":"e3bc46e9-c2d9-4ca7-a005-b8171900903c","resolution":{"observed_at":"2026-08-07T05:11:19.932135Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:11:20.179588Z","title":"Neuro-symbolic concept learner: Interpreting scenes, words, and sentences from nat- ural supervision","venue":null,"work_id":"0a8d81b6-aaae-4f71-8158-cf8e44e7388f","year":2019},"citing_paper":{"arxiv_id":"2506.08553","last_updated":"2025-06-10T08:21:38Z","snapshot_observed_at":"2026-08-15T04:48:41.491398Z","submitted_at":"2025-06-10T08:21:38Z","title":"From Pixels to Graphs: using Scene and Knowledge Graphs for HD-EPIC VQA Challenge","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T05:11:19.936145Z"},"links":{"citing_paper":"/paper/2506.08553"},"observation_digest":"sha256:7228f84556f936bfb6b58cb000e8af10b6a2558f09c7ad586de1f7cf4c1e4678","observation_id":"f794b1ba-7f42-464f-a881-e32cc9fb03c8","resolution":{"observed_at":"2026-08-07T05:11:20.183469Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:11:20.168228Z","title":"Augmented common- sense knowledge for remote object grounding","venue":null,"work_id":"455ee10b-43a7-4564-861e-3d24d695ea73","year":2024},"citing_paper":{"arxiv_id":"2506.08553","last_updated":"2025-06-10T08:21:38Z","snapshot_observed_at":"2026-08-15T04:48:41.491398Z","submitted_at":"2025-06-10T08:21:38Z","title":"From Pixels to Graphs: using Scene and Knowledge Graphs for HD-EPIC VQA Challenge","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T05:11:19.939700Z"},"links":{"citing_paper":"/paper/2506.08553"},"observation_digest":"sha256:46b7be327e6c0d80cb5147ec6ca5f6cae2c19231bd1415b338cd5721da7478ae","observation_id":"e30a909d-c479-445e-8313-a936e9425e0d","resolution":{"observed_at":"2026-08-07T05:11:20.172186Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:11:20.156600Z","title":"Towards unbiased and ro- bust spatio-temporal scene graph generation and anticipa- tion, 2025","venue":null,"work_id":"52d05d8c-f9d7-44be-b860-a2c6e908e735","year":2025},"citing_paper":{"arxiv_id":"2506.08553","last_updated":"2025-06-10T08:21:38Z","snapshot_observed_at":"2026-08-15T04:48:41.491398Z","submitted_at":"2025-06-10T08:21:38Z","title":"From Pixels to Graphs: using Scene and Knowledge Graphs for HD-EPIC VQA Challenge","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T05:11:19.943784Z"},"links":{"citing_paper":"/paper/2506.08553"},"observation_digest":"sha256:3e3da49a7ef56d33a0f3003999f7fca5ead2eb4856a8129549b4147e13b27304","observation_id":"3a6262cb-d8cc-4fd7-8038-7e5750166e15","resolution":{"observed_at":"2026-08-07T05:11:20.160515Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:11:20.145515Z","title":"Pedregosa, G","venue":null,"work_id":"7a3ea0d1-8f24-4996-9da8-6a1408c46bac","year":2011},"citing_paper":{"arxiv_id":"2506.08553","last_updated":"2025-06-10T08:21:38Z","snapshot_observed_at":"2026-08-15T04:48:41.491398Z","submitted_at":"2025-06-10T08:21:38Z","title":"From Pixels to Graphs: using Scene and Knowledge Graphs for HD-EPIC VQA Challenge","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T05:11:19.947463Z"},"links":{"citing_paper":"/paper/2506.08553"},"observation_digest":"sha256:57f011abbd41c31d3e7f6fa1bed50d58856934297ff24c8a48347e50c9a51ffa","observation_id":"2576c7b1-7f67-475a-8946-afb9727e12cd","resolution":{"observed_at":"2026-08-07T05:11:20.149184Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:11:20.133446Z","title":"Sentence-bert: Sentence embeddings using siamese bert-networks","venue":null,"work_id":"64aa49f6-c4a5-4a47-b18d-3b4c84e12917","year":2019},"citing_paper":{"arxiv_id":"2506.08553","last_updated":"2025-06-10T08:21:38Z","snapshot_observed_at":"2026-08-15T04:48:41.491398Z","submitted_at":"2025-06-10T08:21:38Z","title":"From Pixels to Graphs: using Scene and Knowledge Graphs for HD-EPIC VQA Challenge","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T05:11:19.951109Z"},"links":{"citing_paper":"/paper/2506.08553"},"observation_digest":"sha256:e4308cb233e498a5752b0530209b993678756e824ebb54468e84e0120fadc710","observation_id":"eaf39c6b-bb7a-49b8-a893-0c96c71070af","resolution":{"observed_at":"2026-08-07T05:11:20.137882Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:11:20.121258Z","title":"Action scene graphs for long- form understanding of egocentric videos, 2023","venue":null,"work_id":"8730632c-17e9-48f3-8cc1-8841749f8afc","year":2023},"citing_paper":{"arxiv_id":"2506.08553","last_updated":"2025-06-10T08:21:38Z","snapshot_observed_at":"2026-08-15T04:48:41.491398Z","submitted_at":"2025-06-10T08:21:38Z","title":"From Pixels to Graphs: using Scene and Knowledge Graphs for HD-EPIC VQA Challenge","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T05:11:19.954515Z"},"links":{"citing_paper":"/paper/2506.08553"},"observation_digest":"sha256:1106053eee5bc850cd4ecd1670fdf2198e992f7115c4da9fabe31e2e1fb36424","observation_id":"b7d9afb5-6413-4072-9713-88931f8cef68","resolution":{"observed_at":"2026-08-07T05:11:20.125175Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:11:20.109182Z","title":"Atomic: An atlas of ma- chine commonsense for if-then reasoning","venue":null,"work_id":"bb9fe888-e709-46a6-822b-5355bc204562","year":2019},"citing_paper":{"arxiv_id":"2506.08553","last_updated":"2025-06-10T08:21:38Z","snapshot_observed_at":"2026-08-15T04:48:41.491398Z","submitted_at":"2025-06-10T08:21:38Z","title":"From Pixels to Graphs: using Scene and Knowledge Graphs for HD-EPIC VQA Challenge","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T05:11:19.959085Z"},"links":{"citing_paper":"/paper/2506.08553"},"observation_digest":"sha256:458f8c762686f0a65173fef8cb37dc2e02efa4f715a6c51752876f60165f34b3","observation_id":"7063de9e-9a9b-43dd-be6c-d82e2d42bdb8","resolution":{"observed_at":"2026-08-07T05:11:20.113042Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:11:20.096141Z","title":"Concept- net 5.5: An open multilingual graph of general knowledge","venue":null,"work_id":"95076658-1a84-423a-85ce-1cc93f7fcb35","year":2017},"citing_paper":{"arxiv_id":"2506.08553","last_updated":"2025-06-10T08:21:38Z","snapshot_observed_at":"2026-08-15T04:48:41.491398Z","submitted_at":"2025-06-10T08:21:38Z","title":"From Pixels to Graphs: using Scene and Knowledge Graphs for HD-EPIC VQA Challenge","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T05:11:19.963013Z"},"links":{"citing_paper":"/paper/2506.08553"},"observation_digest":"sha256:661771353d976dd29d8cd80e2230a2ed586dad8d9bf8273094d88d9738843c14","observation_id":"eb06ebe3-fb83-4fbe-a318-3a0aee9005a0","resolution":{"observed_at":"2026-08-07T05:11:20.100797Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:11:20.082979Z","title":"Concept- net 5.5: An open multilingual graph of general knowledge,","venue":null,"work_id":"1fa71898-f162-4ead-8641-dc0df9d974d6","year":null},"citing_paper":{"arxiv_id":"2506.08553","last_updated":"2025-06-10T08:21:38Z","snapshot_observed_at":"2026-08-15T04:48:41.491398Z","submitted_at":"2025-06-10T08:21:38Z","title":"From Pixels to Graphs: using Scene and Knowledge Graphs for HD-EPIC VQA Challenge","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T05:11:19.966591Z"},"links":{"citing_paper":"/paper/2506.08553"},"observation_digest":"sha256:75fec52291eed2bc64cbc29e4143382c3682c0b6c9f0dd68cecaf10a7235b8b8","observation_id":"61b922e8-d390-45f0-918f-9cc63241f79a","resolution":{"observed_at":"2026-08-07T05:11:20.087645Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:11:20.070987Z","title":"Learning situation hyper-graphs for video question answer- ing","venue":null,"work_id":"f89710b3-ce68-4fb1-a3ad-5ab120b611e9","year":null},"citing_paper":{"arxiv_id":"2506.08553","last_updated":"2025-06-10T08:21:38Z","snapshot_observed_at":"2026-08-15T04:48:41.491398Z","submitted_at":"2025-06-10T08:21:38Z","title":"From Pixels to Graphs: using Scene and Knowledge Graphs for HD-EPIC VQA Challenge","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T05:11:19.970671Z"},"links":{"citing_paper":"/paper/2506.08553"},"observation_digest":"sha256:d7aa621eba79bc97979e3277160f799aadd6d5e3e3cefce836595d3b892e8c7b","observation_id":"26dc9ebc-4dac-454d-b468-42cee761571f","resolution":{"observed_at":"2026-08-07T05:11:20.074898Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:11:20.059058Z","title":"Scene graph generation by iterative message passing","venue":null,"work_id":"54881aea-9e9c-4a24-b4d4-2f0418c1a241","year":2017},"citing_paper":{"arxiv_id":"2506.08553","last_updated":"2025-06-10T08:21:38Z","snapshot_observed_at":"2026-08-15T04:48:41.491398Z","submitted_at":"2025-06-10T08:21:38Z","title":"From Pixels to Graphs: using Scene and Knowledge Graphs for HD-EPIC VQA Challenge","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T05:11:19.975080Z"},"links":{"citing_paper":"/paper/2506.08553"},"observation_digest":"sha256:428a1316b1cafdede6e0797f6b4eb48d33908aaddea5ec9cf6ea7cf0e7531c87","observation_id":"4f776e03-62e6-4fe9-b272-7deba1911577","resolution":{"observed_at":"2026-08-07T05:11:20.062960Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:11:20.046048Z","title":"Neuro-symbolic visual reasoning: Disentangling ”vi- sual” from ”reasoning”","venue":null,"work_id":"b25c1521-1c03-46bd-8403-d9d78db28ad2","year":2018},"citing_paper":{"arxiv_id":"2506.08553","last_updated":"2025-06-10T08:21:38Z","snapshot_observed_at":"2026-08-15T04:48:41.491398Z","submitted_at":"2025-06-10T08:21:38Z","title":"From Pixels to Graphs: using Scene and Knowledge Graphs for HD-EPIC VQA Challenge","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T05:11:19.978859Z"},"links":{"citing_paper":"/paper/2506.08553"},"observation_digest":"sha256:c668c6cd7858afec14e75f57f6a718aaaf89a7d4c28712a082ea1b0b74a14fd0","observation_id":"957fbb91-98dc-4674-b2db-2fe9e6a755cd","resolution":{"observed_at":"2026-08-07T05:11:20.050199Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:11:20.031917Z","title":"Jasper and stella: distillation of sota embedding models,","venue":null,"work_id":"460e70db-0c53-4371-a93f-3c0881b853a1","year":null},"citing_paper":{"arxiv_id":"2506.08553","last_updated":"2025-06-10T08:21:38Z","snapshot_observed_at":"2026-08-15T04:48:41.491398Z","submitted_at":"2025-06-10T08:21:38Z","title":"From Pixels to Graphs: using Scene and Knowledge Graphs for HD-EPIC VQA Challenge","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T05:11:19.983039Z"},"links":{"citing_paper":"/paper/2506.08553"},"observation_digest":"sha256:35c4216e2a4634e6a331c28ac94c82cde49c6f59d43477b093f784a99fd89951","observation_id":"692377ea-c82e-4a12-bfc1-c2e2479fe93c","resolution":{"observed_at":"2026-08-07T05:11:20.037717Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2506.08553","last_updated":"2025-06-10T08:21:38Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-15T04:48:41.491398Z","submitted_at":"2025-06-10T08:21:38Z","title":"From Pixels to Graphs: using Scene and Knowledge Graphs for HD-EPIC VQA Challenge"},"reference_resolution":{"displayed":27,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":4,"verified_exact":0,"verified_fuzzy":23},"total_outbound_references":27},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"thesis":"As of 15 August 2026, this Paper Citation Record lists 27 of 27 outbound references and 1 inbound Pith citation observation for arXiv:2506.08553."}