{"as_of":"2026-08-17T13:49:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:dded3635f3f193ffb4f919756acb47ed0c222a2260422710db027c47cf5fe6c2","coverage":[{"denominator":60,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":60,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T23:50:44.081050Z","state":"measured"},{"denominator":60,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":60,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-17T06:30:58.91139+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2505.03581/citation-record","integrity":"/paper/2505.03581/integrity","json":"/paper/2505.03581/citation-record.json","paper":"/paper/2505.03581"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.706282Z","title":"Conceptgraphs: Open-vocabulary 3d scene graphs for perception and planning,","venue":null,"work_id":"7cdf7201-33a8-4a99-a6db-41ada5dfd4fc","year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.879396Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:7cd149bd92ce454bea75a8b66f0ea0119a9adb1f389629e899d4b277c42a8188","observation_id":"2f0771a3-f1ca-4c86-8055-a9b79bb5f917","resolution":{"observed_at":"2026-08-15T23:50:44.709445Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.696442Z","title":"Beyond bare queries: Open-vocabulary object retrieval with 3d scene graph,","venue":null,"work_id":"e23eaadb-e604-4e1c-8ce8-25d7ef75c582","year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.883967Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:982163a4d833b2db81a0024e86ce14c9a4e9ff53e2efbc38f2468bf7dd42cfce","observation_id":"4c7416d5-173b-4e37-be71-32794146e0e3","resolution":{"observed_at":"2026-08-15T23:50:44.700268Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.685119Z","title":"Search3d: Hierarchical open-vocabulary 3d segmenta- tion,","venue":null,"work_id":"de22d83e-d791-400a-85b5-0b22bd15f883","year":2025},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.888014Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:705c78a10f5fa06c896418a852989ed328d5c8474ac4f2cef8f28981a407aef1","observation_id":"84232814-3510-4da1-a4f1-dd96325f2251","resolution":{"observed_at":"2026-08-15T23:50:44.689789Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:43.891225Z","title":"Hierarchical open-vocabulary 3d scene graphs for language-grounded robot navigation,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.891225Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:966f92cd604dd07814e1e492b0878946f63706832cf6f7e0713f46af0e5bf10d","observation_id":"d7074e35-bd7b-4e54-99cb-ece5e1f239d7","resolution":{"observed_at":"2026-08-15T23:50:43.891225Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:43.895182Z","title":"Clio: Real-time task-driven open-set 3d scene graphs,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.895182Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:8a962f6434aad18d75424f294a76b70e9b91c6fc7702f4476605608400d93b7e","observation_id":"b7d94e18-e615-42c0-b8b0-51c029add4f6","resolution":{"observed_at":"2026-08-15T23:50:43.895182Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.664806Z","title":"4d panoptic scene graph generation,","venue":null,"work_id":"98b0287e-2bc6-4574-ba16-5b0f9b21a446","year":2023},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.898419Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:ba933df6ca89aedd528c78988cb45739fe95efea82d6ae92426789432737a680","observation_id":"b9494be8-c440-42a5-b3bf-6204d5c977cd","resolution":{"observed_at":"2026-08-15T23:50:44.668082Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.654878Z","title":"G-retriever: Retrieval-augmented generation for textual graph understanding and question answering,","venue":null,"work_id":"0681acce-bcba-45fe-b285-8c465ab26894","year":2025},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.902087Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:5f41278aa6ab85c1862c08aa17b45cb4f4ace8e45bcb4ebecda25e9188d48e68","observation_id":"46c15d20-7df7-4f50-a137-3da68dc4bdce","resolution":{"observed_at":"2026-08-15T23:50:44.658325Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05862","last_updated":"2024-02-08T17:51:44Z","snapshot_observed_at":"2026-08-16T14:20:08.727085Z","submitted_at":"2024-02-08T17:51:44Z","title":"Let Your Graph Do the Talking: Encoding Structured Data for LLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05862","snapshot_observed_at":"2026-08-15T23:50:43.905097Z","title":"Let your graph do the talking: Encoding structured data for llms,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.905097Z"},"links":{"cited_paper":"/paper/2402.05862","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:986642f96a646869e3f1c1ecbed5df711e8c97b28bad98c16eca711c6df8fcb2","observation_id":"645bd11b-dc08-4e04-8761-17006d2928db","resolution":{"observed_at":"2026-08-15T23:50:43.905097Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.643891Z","title":"Can llms enhance performance prediction for deep learning models?","venue":null,"work_id":"cfcbcc08-9f4c-4fab-9731-aaaaec3ffd0a","year":null},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.908611Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:0806e5ba2158dd32cb756ad14eb496810d69ae76266999573ff3286d4c2ebf5b","observation_id":"e8cc9849-0fb3-46d9-bcf3-73b8e8b9bccb","resolution":{"observed_at":"2026-08-15T23:50:44.647941Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.09711","last_updated":"2024-05-15T21:53:54Z","snapshot_observed_at":"2026-08-16T21:19:34.047707Z","submitted_at":"2024-05-15T21:53:54Z","title":"STAR: A Benchmark for Situated Reasoning in Real-World Videos","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.09711","snapshot_observed_at":"2026-08-15T23:50:43.912140Z","title":"Star: A benchmark for situated reasoning in real-world videos,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.912140Z"},"links":{"cited_paper":"/paper/2405.09711","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:ae0d57e10be940829bf529773a9e7dfef31d8dae20f83dcdb101ccc45b567c22","observation_id":"8128cca7-1eee-44b0-a647-76d99cacdfdf","resolution":{"observed_at":"2026-08-15T23:50:43.912140Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.634277Z","title":"Agqa 2.0: An updated benchmark for compositional spatio-temporal reasoning,","venue":null,"work_id":"59d061a8-a3a2-4f0a-9328-8a1aa8421ddd","year":null},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.915839Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:81d84ccf8502c8fb5c9d440aab4104ba73b1ee8b9af5f1ad7b4238115088b0d9","observation_id":"8c57e5aa-525e-46ac-9490-95e8720d9733","resolution":{"observed_at":"2026-08-15T23:50:44.637619Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:43.922505Z","title":"Visual genome: Connecting language and vision using crowdsourced dense image annotations,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.922505Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:6ee9b4885410bb98a511ac6907a56b7ab01665e34156a4b7954526fd3bfaefdd","observation_id":"1885515b-95f2-4100-8240-6e09a118c67a","resolution":{"observed_at":"2026-08-15T23:50:43.922505Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:43.925544Z","title":"Gqa: A new dataset for real- world visual reasoning and compositional question answering,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.925544Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:c1e6db1a869d84bf6d545c75b1affdb1e1827f697214bba21a96b240c8ad29e6","observation_id":"f758b0d9-18c6-47ca-b6e9-be1bb9780696","resolution":{"observed_at":"2026-08-15T23:50:43.925544Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.615326Z","title":"Panoptic scene graph generation,","venue":null,"work_id":"18dd88f2-1a14-4876-8049-072983bc882a","year":2022},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.928530Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:62c230524b138000a4af45635653ff8bce807c809cfba938b22faabc1b09542c","observation_id":"9a767cad-a860-4189-86b5-bef289d6efa8","resolution":{"observed_at":"2026-08-15T23:50:44.618448Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.606047Z","title":"Action genome: Actions as compositions of spatio-temporal scene graphs,","venue":null,"work_id":"b1cd20d0-f496-4d4b-a713-5972178e6c81","year":2020},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.931740Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:b74ec7c5f611399c79a1dc3257dc49cca6308d82a489ed45aa9d62970812ec0a","observation_id":"65521ffd-71c5-42e0-b747-d37d6c798f85","resolution":{"observed_at":"2026-08-15T23:50:44.609510Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.596888Z","title":"Panoptic video scene graph generation,","venue":null,"work_id":"a98a2655-3b38-4449-9ff6-dc2be15dc72b","year":2023},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.934810Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:329f5d2133279c73dfd34438f2204a441e48a5e1be71fb637996141436c37839","observation_id":"0fed5c8e-ac80-492e-a37b-dd2684cfbfa4","resolution":{"observed_at":"2026-08-15T23:50:44.600160Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:43.938041Z","title":"Egtr: Extracting graph from transformer for scene graph generation,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.938041Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:7c6c9523ceff0b889194c8c520634b760546a5fbf2fab01ba1732ef31c2fcc12","observation_id":"a090021c-ddbf-4edd-b9a9-393dcf6de988","resolution":{"observed_at":"2026-08-15T23:50:43.938041Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:43.941400Z","title":"Reltr: Relation transformer for scene graph generation,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.941400Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:6e6345ec6732b1092185bc42f827ab148ee27019aca7e66802d95eae5e327b31","observation_id":"30a67c2d-1411-459d-a581-f32df200fe56","resolution":{"observed_at":"2026-08-15T23:50:43.941400Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.577043Z","title":"Oed: towards one-stage end-to- end dynamic scene graph generation,","venue":null,"work_id":"b337302e-c865-48f8-b514-401429587e0e","year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.945053Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:ebc7040d41e3beff1438d021527b0ac1b6b1aaf24b1864bcf8c59367ba8f1232","observation_id":"a6806e38-1f43-436e-a43a-bf1604f5a723","resolution":{"observed_at":"2026-08-15T23:50:44.580376Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00714","last_updated":"2024-10-28T16:37:57Z","snapshot_observed_at":"2026-07-06T18:55:41.459417Z","submitted_at":"2024-08-01T17:00:08Z","title":"SAM 2: Segment Anything in Images and Videos","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.00714","snapshot_observed_at":"2026-08-15T23:50:43.948634Z","title":"Sam 2: Segment anything in images and videos,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.948634Z"},"links":{"cited_paper":"/paper/2408.00714","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:27db8eff5cc557bef0e2f0f0732fa06d243da110480b9057320520f51ab1230c","observation_id":"1605903d-9619-4b17-b3b2-2cfbf2bf3ee1","resolution":{"observed_at":"2026-08-15T23:50:43.948634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:43.952811Z","title":"Llavanext: Improved reasoning, ocr, and world knowledge,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.952811Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:64d55d48637f671248e4d92305007ea68cf7efe397088ea914a3453630f7c4c7","observation_id":"7e316375-8fa3-404a-83ff-1832345ffe1f","resolution":{"observed_at":"2026-08-15T23:50:43.952811Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:43.956865Z","title":"Yolo-world: Real-time open-vocabulary object detection,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.956865Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:76e36d84cc3c06ac034a936f3d1ae126ea0539f867fd2153b679df7ba349eed4","observation_id":"22188316-e050-47d6-9f6f-be776b314683","resolution":{"observed_at":"2026-08-15T23:50:43.956865Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-17T09:58:46.058102Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-15T23:50:43.960371Z","title":"Gpt-4 technical report,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.960371Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:7e7fda200d0c714b992da2f0e71006ddc9edaf2125998d5e04e9174a91e82c03","observation_id":"e3418715-8ace-498d-bcf7-ef9b3d605286","resolution":{"observed_at":"2026-08-15T23:50:43.960371Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04652","last_updated":"2025-01-21T10:12:05Z","snapshot_observed_at":"2026-08-17T07:57:04.845849Z","submitted_at":"2024-03-07T16:52:49Z","title":"Yi: Open Foundation Models by 01.AI","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04652","snapshot_observed_at":"2026-08-15T23:50:43.963321Z","title":"Yi: Open foundation models by 01. ai,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.963321Z"},"links":{"cited_paper":"/paper/2403.04652","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:4e33e2434e324d83c6ed9c1be96c28b13b2ae035a85066d7f8e52d6ea1cbc7a5","observation_id":"5c725269-06a3-4ead-ae5e-0a3f44765277","resolution":{"observed_at":"2026-08-15T23:50:43.963321Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.04468","last_updated":"2026-04-25T07:16:42Z","snapshot_observed_at":"2026-08-03T08:48:57.969106Z","submitted_at":"2024-12-05T18:59:55Z","title":"NVILA: Efficient Frontier Visual Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.04468","snapshot_observed_at":"2026-08-15T23:50:43.967414Z","title":"Nvila: Efficient frontier visual language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.967414Z"},"links":{"cited_paper":"/paper/2412.04468","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:69c580a6f73de2299e5c7db9986ad2396f331fda181cc1a56cce961a099196dd","observation_id":"77f7d309-02d3-4c7c-b892-64d46f8a505c","resolution":{"observed_at":"2026-08-15T23:50:43.967414Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.13923","last_updated":"2025-02-19T18:00:14Z","snapshot_observed_at":"2026-08-14T04:17:22.593941Z","submitted_at":"2025-02-19T18:00:14Z","title":"Qwen2.5-VL Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.13923","snapshot_observed_at":"2026-08-15T23:50:43.970524Z","title":"Qwen2. 5-vl technical report,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.970524Z"},"links":{"cited_paper":"/paper/2502.13923","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:18f7347ce0a047c40e10bd6d6702302cae4122ff1bda16d739c262acd379cfa7","observation_id":"740d89b9-7117-4b46-a85b-ff8648391f89","resolution":{"observed_at":"2026-08-15T23:50:43.970524Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.06723","last_updated":"2025-02-26T22:54:53Z","snapshot_observed_at":"2026-08-16T13:35:50.956800Z","submitted_at":"2024-07-09T09:55:04Z","title":"Graph-Based Captioning: Enhancing Visual Descriptions by Interconnecting Region Captions","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.06723","snapshot_observed_at":"2026-08-15T23:50:43.973960Z","title":"Graph- based captioning: Enhancing visual descriptions by interconnecting region captions,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.973960Z"},"links":{"cited_paper":"/paper/2407.06723","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:cd57a904143ac5983362f021393e0d1efcfae51efa7c9d14dfcccaa0494646f7","observation_id":"8a8e3370-8c8f-4ecf-ad26-162d9a30876b","resolution":{"observed_at":"2026-08-15T23:50:43.973960Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.07012","last_updated":"2024-12-29T03:52:23Z","snapshot_observed_at":"2026-08-16T13:00:22.377701Z","submitted_at":"2024-12-09T21:44:02Z","title":"ProVision: Programmatically Scaling Vision-centric Instruction Data for Multimodal Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.07012","snapshot_observed_at":"2026-08-15T23:50:43.977587Z","title":"Provision: Programmatically scaling vision-centric instruction data for multimodal language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.977587Z"},"links":{"cited_paper":"/paper/2412.07012","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:f8228f319e88f26f6efdc2363c798cd0e71bf69f4665a8492931a82501ea6514","observation_id":"04a36963-6c86-4e53-87ab-7c9073ebd39a","resolution":{"observed_at":"2026-08-15T23:50:43.977587Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.557312Z","title":"Llm4sgg: large language models for weakly supervised scene graph generation,","venue":null,"work_id":"658dfab4-3759-43aa-922f-47664282bb9d","year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.980913Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:74e9bee3b734e846b33c14b3de187d3a3d542c5533558ecaa712b158826d3d6a","observation_id":"f6eeaa08-33b7-48ca-b31f-e80b1c3944c1","resolution":{"observed_at":"2026-08-15T23:50:44.560483Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.14171","last_updated":"2025-07-02T21:00:36Z","snapshot_observed_at":"2026-08-14T22:29:28.847917Z","submitted_at":"2024-12-18T18:59:54Z","title":"Thinking in Space: How Multimodal Large Language Models See, Remember, and Recall Spaces","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.14171","snapshot_observed_at":"2026-08-15T23:50:43.984048Z","title":"Thinking in space: How multimodal large language models see, remember, and recall spaces,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.984048Z"},"links":{"cited_paper":"/paper/2412.14171","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:367a70e9dcde3e63e92010361a050907283959424d6d96a8819868ac79ca92b5","observation_id":"93d64783-f5cc-43c8-b905-59f1660eccb1","resolution":{"observed_at":"2026-08-15T23:50:43.984048Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.18938","last_updated":"2024-12-03T03:56:52Z","snapshot_observed_at":"2026-08-16T13:14:41.487643Z","submitted_at":"2024-09-27T17:38:36Z","title":"From Seconds to Hours: Reviewing MultiModal Large Language Models on Comprehensive Long Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.18938","snapshot_observed_at":"2026-08-15T23:50:43.987304Z","title":"From seconds to hours: Reviewing multimodal large language models on comprehensive long video understanding,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.987304Z"},"links":{"cited_paper":"/paper/2409.18938","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:cd1c0c3848570da0ad9d57500d6237f647d0d616f8550f8c6e80545820f93465","observation_id":"db1551b9-ffdd-478d-b7fb-7678a1a290c6","resolution":{"observed_at":"2026-08-15T23:50:43.987304Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.02765","last_updated":"2025-01-06T05:15:59Z","snapshot_observed_at":"2026-08-14T12:32:34.328935Z","submitted_at":"2025-01-06T05:15:59Z","title":"Visual Large Language Models for Generalized and Specialized Applications","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.02765","snapshot_observed_at":"2026-08-15T23:50:43.990891Z","title":"Visual large language models for generalized and specialized applications,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.990891Z"},"links":{"cited_paper":"/paper/2501.02765","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:27c373f935b9c1c2b2aeee578a8acb78dc354cc6846506dab7466eb5b8e0eb9d","observation_id":"9818144f-18fa-476e-a85b-1e19b57e3fa5","resolution":{"observed_at":"2026-08-15T23:50:43.990891Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.547692Z","title":"(2.5+ 1) d spatio- temporal scene graphs for video question answering,","venue":null,"work_id":"c677c7d2-f5a9-48db-a378-62b1d7c2db90","year":2022},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.993936Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:c55e033d40dd347debe1720620b3c5150be7f71db5ed44cfa1350928740acfca","observation_id":"3d0cf3e8-42d0-4d39-a3f8-79333589329e","resolution":{"observed_at":"2026-08-15T23:50:44.551272Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.538485Z","title":"Action scene graphs for long-form understanding of egocentric videos,","venue":null,"work_id":"b58772ee-5ec9-413f-b2c4-9de58fe5f38f","year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.997532Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:c87f1626555d3644de6712abf2bd80f322aa12049e1512cb8c83c5040739c849","observation_id":"4cc7d048-e096-4191-b2f3-254a8c9b05f7","resolution":{"observed_at":"2026-08-15T23:50:44.541815Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.18042","last_updated":"2025-03-31T08:16:49Z","snapshot_observed_at":"2026-08-16T07:06:08.115739Z","submitted_at":"2024-11-27T04:24:39Z","title":"HyperGLM: HyperGraph for Video Scene Graph Generation and Anticipation","version":2},"cited_work":{"arxiv_id":"2411.18042","doi":null,"metadata_source":"pith","pith_arxiv_id":"2411.18042","snapshot_observed_at":"2026-08-15T23:50:44.314610Z","title":"HyperGLM: HyperGraph for Video Scene Graph Generation and Anticipation","venue":"cs.CV","work_id":"61c5b646-5432-4e8d-923b-74b8ac091cbc","year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:44.000451Z"},"links":{"cited_paper":"/paper/2411.18042","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:fcf75ccb889639ad7a14bab42569dae9a1d492ef08d956c79f376a4faefe4982","observation_id":"1b858b1f-bfeb-47f1-b0bd-5e31f8284cb9","resolution":{"observed_at":"2026-08-15T23:50:44.318336Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.00161","last_updated":"2025-03-30T14:31:41Z","snapshot_observed_at":"2026-08-16T04:20:19.899128Z","submitted_at":"2024-11-29T11:54:55Z","title":"STEP: Enhancing Video-LLMs' Compositional Reasoning by Spatio-Temporal Graph-guided Self-Training","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.00161","snapshot_observed_at":"2026-08-15T23:50:44.003979Z","title":"Step: Enhancing video-llms’ composi- tional reasoning by spatio-temporal graph-guided self-training,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:44.003979Z"},"links":{"cited_paper":"/paper/2412.00161","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:895261cd74b5c0ea03b3ebace49ed68e1fafcb56f42c28de3655915323319298","observation_id":"d0f2412a-688e-4361-b79c-5c67e17a3d8c","resolution":{"observed_at":"2026-08-15T23:50:44.003979Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.13663","last_updated":"2024-12-19T06:32:26Z","snapshot_observed_at":"2026-08-16T17:36:35.256420Z","submitted_at":"2024-12-18T09:39:44Z","title":"Smarter, Better, Faster, Longer: A Modern Bidirectional Encoder for Fast, Memory Efficient, and Long Context Finetuning and Inference","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.13663","snapshot_observed_at":"2026-08-15T23:50:44.007182Z","title":"Smarter, better, faster, longer: A modern bidirectional encoder for fast, memory efficient, and long context finetuning and inference,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:44.007182Z"},"links":{"cited_paper":"/paper/2412.13663","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:7bd14752f59a25151956a926d48539056221a6d237ced4ae93597448c7eb5c96","observation_id":"a1abcacc-c6b9-435a-a869-a612ea80fcd2","resolution":{"observed_at":"2026-08-15T23:50:44.007182Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.010411Z","title":"Benchmarking graph neural networks,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:44.010411Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:81e897ed93ddd3a6165d67804c46597ac5e1736bab06639d05a625558630e607","observation_id":"d4250c15-2b0a-4a16-ae60-b28dee74ec5e","resolution":{"observed_at":"2026-08-15T23:50:44.010411Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2009.03509","last_updated":"2021-05-10T02:23:20Z","snapshot_observed_at":"2026-08-11T19:52:55.524234Z","submitted_at":"2020-09-08T04:04:04Z","title":"Masked Label Prediction: Unified Message Passing Model for Semi-Supervised Classification","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2009.03509","snapshot_observed_at":"2026-08-15T23:50:44.013299Z","title":"Masked label prediction: Unified message passing model for semi-supervised classification,","venue":null,"work_id":null,"year":2009},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:44.013299Z"},"links":{"cited_paper":"/paper/2009.03509","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:284ca352650a821e9b5ee409a088b3de94e94161c90e0a95894fbd4148d6e86c","observation_id":"d938cc35-c39c-4172-ac51-f9f679921791","resolution":{"observed_at":"2026-08-15T23:50:44.013299Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.016593Z","title":"Roformer: En- hanced transformer with rotary position embedding,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:44.016593Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:1faea35f0eecc40919d00b562d377857c1e289d806830d6a23482e580f385b82","observation_id":"bd3c880a-c483-4c30-9183-6ac3d7a94fd0","resolution":{"observed_at":"2026-08-15T23:50:44.016593Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.019556Z","title":"Blip-2: Bootstrapping language- image pre-training with frozen image encoders and large language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:44.019556Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:5f40dc0ff67beb2a7528b45e86ca404bd1064932c8fddfa63e18670f23eac638","observation_id":"a12445ac-7fcb-4ea8-88e4-d207e0a54489","resolution":{"observed_at":"2026-08-15T23:50:44.019556Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.512868Z","title":"Agqa: A benchmark for compositional spatio-temporal reasoning,","venue":null,"work_id":"b95d31ac-8819-447f-9290-db34fcdb0126","year":2021},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:44.022635Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:7a9bddb019fe7f5e72eab9fcd3ed35063ec6c0c5fbc79e0f501cbee8d84aa87c","observation_id":"8cd2c11b-fd18-4e51-806c-f2e36ee7cb5b","resolution":{"observed_at":"2026-08-15T23:50:44.516269Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.025383Z","title":"Lora: Low-rank adaptation of large language models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:44.025383Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:6951fbfe4bd149b9a506d39c1f3a70f16e30c63d869655d0897b8558225bf9b3","observation_id":"c168e277-342d-4817-9a93-37fe007677cd","resolution":{"observed_at":"2026-08-15T23:50:44.025383Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-08-13T17:20:44.002518Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-15T23:50:44.029059Z","title":"The llama 3 herd of models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:44.029059Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:465025f6ef080cdbaf3da2fe7e038f8e10a4948442f7828b0121d70755649fe7","observation_id":"c8f94f6c-05ae-411a-8abb-bfb10691e8ad","resolution":{"observed_at":"2026-08-15T23:50:44.029059Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.11636","last_updated":"2023-02-22T20:24:35Z","snapshot_observed_at":"2026-08-16T15:53:26.353974Z","submitted_at":"2023-02-22T20:24:35Z","title":"Do We Really Need Complicated Model Architectures For Temporal Networks?","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.11636","snapshot_observed_at":"2026-08-15T23:50:44.032856Z","title":"Do we really need complicated model architectures for temporal networks?","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:44.032856Z"},"links":{"cited_paper":"/paper/2302.11636","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:55e4b39221bc39469d147ca2f5a87583ed8ad295ef7509c0a7aed2c9c2ee3981","observation_id":"85b94c4e-2a7b-44a0-9418-7c43271d08d7","resolution":{"observed_at":"2026-08-15T23:50:44.032856Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.036476Z","title":"Attention is all you need,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:44.036476Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:7b17fd606fdeed22bf4d7eb061fdfdd1e26e6a741fc67b19177b232c875549c4","observation_id":"fc2ae432-b86d-49e9-82a7-11240a495aa5","resolution":{"observed_at":"2026-08-15T23:50:44.036476Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.10698","last_updated":"2024-07-21T00:42:15Z","snapshot_observed_at":"2026-08-16T14:17:58.795960Z","submitted_at":"2024-02-16T13:59:07Z","title":"Question-Instructed Visual Descriptions for Zero-Shot Video Question Answering","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.10698","snapshot_observed_at":"2026-08-15T23:50:44.040102Z","title":"Question-instructed visual descrip- tions for zero-shot video question answering,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:44.040102Z"},"links":{"cited_paper":"/paper/2402.10698","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:ac84d2daf4b7f307a2207179ca233724a7338f7557ebfd24d63629d58aa0bab8","observation_id":"ea23b79a-d636-4a5f-9585-088fc0a12db6","resolution":{"observed_at":"2026-08-15T23:50:44.040102Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.491931Z","title":"Mist: Multi-modal iterative spatial-temporal transformer for long-form video question answering,","venue":null,"work_id":"59b9200d-f72e-429d-a0fe-c74849105422","year":2023},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:44.044120Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:0d816c0f55be1d7d1fa64e8ab0958d3b9d43d4f4ee0e4392ef95edfed5b1cd34","observation_id":"f3d640f6-5953-462d-85d4-55aaab99866f","resolution":{"observed_at":"2026-08-15T23:50:44.495603Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.482131Z","title":"Self-chained image-language model for video localization and question answering,","venue":null,"work_id":"91c3f047-6d9d-47ad-8570-74071b190ffb","year":2023},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:44.047370Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:e7e7b8526e004cdc6206bc368d408ab52e556c238068fa539fd3e41536907720","observation_id":"41a135df-9a6f-469b-b794-3ed9cdc277a5","resolution":{"observed_at":"2026-08-15T23:50:44.485545Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.472297Z","title":"Vila: Efficient video-language alignment for video question answering,","venue":null,"work_id":"516e9211-7cd3-4b75-adc6-84796e000e01","year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:44.051426Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:0bb8ff38743ad3b03cfe269f3567c19bba7988541fb5ca0b546b32ed14af2596","observation_id":"219cc9f7-2e91-40e7-a4f4-2e2d071e4456","resolution":{"observed_at":"2026-08-15T23:50:44.476300Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15047","last_updated":"2024-07-23T14:56:22Z","snapshot_observed_at":"2026-08-16T13:32:25.418447Z","submitted_at":"2024-07-21T04:09:37Z","title":"End-to-End Video Question Answering with Frame Scoring Mechanisms and Adaptive Sampling","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15047","snapshot_observed_at":"2026-08-15T23:50:44.054394Z","title":"End- to-end video question answering with frame scoring mechanisms and adaptive sampling,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:44.054394Z"},"links":{"cited_paper":"/paper/2407.15047","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:ad3b734575668c594c69239a7298e94107dc16bd82672589f2c0cdd679393823","observation_id":"17ca962d-91c5-44f7-ae3e-ab8089b6aac4","resolution":{"observed_at":"2026-08-15T23:50:44.054394Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.17778","last_updated":"2024-01-22T00:54:30Z","snapshot_observed_at":"2026-08-16T15:19:40.403833Z","submitted_at":"2023-06-30T16:31:14Z","title":"Look, Remember and Reason: Grounded reasoning in videos with language models","version":3},"cited_work":{"arxiv_id":"2306.17778","doi":null,"metadata_source":"pith","pith_arxiv_id":"2306.17778","snapshot_observed_at":"2026-08-15T23:50:44.154476Z","title":"Look, Remember and Reason: Grounded reasoning in videos with language models","venue":"cs.CV","work_id":"6ae677c9-b8db-427a-82b0-8df004e71e7b","year":2023},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:44.057798Z"},"links":{"cited_paper":"/paper/2306.17778","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:28bd00515ce92117b68c593bc434deebf5bc6602b6fca73ab9cde5e5a188efff","observation_id":"9c57fd8c-d894-4d96-baa6-3b39df3c6587","resolution":{"observed_at":"2026-08-15T23:50:44.244898Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.462389Z","title":"Glance and focus: Memory prompt- ing for multi-event video question answering,","venue":null,"work_id":"df8b41b3-24b6-4b7a-a8d8-969015599bef","year":2023},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:44.060839Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:f659ed7109f203b08be20b1e9f78f4b5e6675876257d16b3477121fdd59d4bef","observation_id":"9fe86359-9402-4674-81ae-90c404ae6f0a","resolution":{"observed_at":"2026-08-15T23:50:44.465775Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.452359Z","title":"Learning to reason iteratively and parallelly for complex visual reasoning scenarios,","venue":null,"work_id":"4eaa3226-fa95-45f8-a704-13ff75331aac","year":2025},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:44.064372Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:d48442a83f92c1c776ae83f37993ad475f7da432f935de6daf9f3a8040a0b7d1","observation_id":"204bab2a-2e42-4642-97bb-685fc96fff48","resolution":{"observed_at":"2026-08-15T23:50:44.456171Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.16050","last_updated":"2024-10-03T09:24:56Z","snapshot_observed_at":"2026-08-16T14:15:34.666227Z","submitted_at":"2024-02-25T10:27:46Z","title":"Efficient Temporal Extrapolation of Multimodal Large Language Models with Temporal Grounding Bridge","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.16050","snapshot_observed_at":"2026-08-15T23:50:44.067588Z","title":"Efficient temporal extrapolation of multimodal large language models with temporal grounding bridge,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:44.067588Z"},"links":{"cited_paper":"/paper/2402.16050","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:bf6780d4f2b14c600b1ade9fcdd9c2cc03719828fb1a077b8bb49fe5e7179bb6","observation_id":"7e94c167-1768-4650-bcc6-10a6cde4d138","resolution":{"observed_at":"2026-08-15T23:50:44.067588Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.03941","last_updated":"2022-10-08T07:03:31Z","snapshot_observed_at":"2026-08-16T16:25:56.512457Z","submitted_at":"2022-10-08T07:03:31Z","title":"Learning Fine-Grained Visual Understanding for Video Question Answering via Decoupling Spatial-Temporal Modeling","version":1},"cited_work":{"arxiv_id":"2210.03941","doi":null,"metadata_source":"pith","pith_arxiv_id":"2210.03941","snapshot_observed_at":"2026-08-15T23:50:44.127936Z","title":"Learning Fine-Grained Visual Understanding for Video Question Answering via Decoupling Spatial-Temporal Modeling","venue":"cs.CV","work_id":"a8096dd3-4b9e-4a75-a3f8-0611e545b3c2","year":2022},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:44.071264Z"},"links":{"cited_paper":"/paper/2210.03941","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:5965ccdfb4a320dd2f5027d342c930cd8bf22dab04640784736898364923b379","observation_id":"fec0ce2d-c820-44ef-b6b5-a7500e15d60f","resolution":{"observed_at":"2026-08-15T23:50:44.134813Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07630","last_updated":"2024-05-27T04:04:40Z","snapshot_observed_at":"2026-08-16T14:19:21.784480Z","submitted_at":"2024-02-12T13:13:04Z","title":"G-Retriever: Retrieval-Augmented Generation for Textual Graph Understanding and Question Answering","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.07630","snapshot_observed_at":"2026-08-15T23:50:44.074558Z","title":"G-retriever: Retrieval-augmented generation for textual graph understanding and question answering,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:44.074558Z"},"links":{"cited_paper":"/paper/2402.07630","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:c6a314cbdfb36007f43fd0e84ec42d7ec894eac3dd89df49802867a1c419820a","observation_id":"e527ce8f-8f68-4236-8d92-491033c3c6e2","resolution":{"observed_at":"2026-08-15T23:50:44.074558Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.441095Z","title":"A note on the prize collecting traveling salesman problem,","venue":null,"work_id":"21500474-8afc-4e46-ac53-173aba94ab26","year":1993},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:44.078054Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:9a067f2fc4ccce9cbc3152392c09c47624bf4dcf19c1373651a9f5f4710cb65e","observation_id":"49832ac9-9274-4e89-9d63-778c489cd185","resolution":{"observed_at":"2026-08-15T23:50:44.445522Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.17497","last_updated":"2023-06-01T04:56:26Z","snapshot_observed_at":"2026-08-16T15:29:02.304597Z","submitted_at":"2023-05-27T15:38:31Z","title":"FACTUAL: A Benchmark for Faithful and Consistent Textual Scene Graph Parsing","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.17497","snapshot_observed_at":"2026-08-15T23:50:44.081050Z","title":"Factual: A benchmark for faithful and consistent textual scene graph parsing,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:44.081050Z"},"links":{"cited_paper":"/paper/2305.17497","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:fdea12372f4a114aecf61b8ce69b32117cfb068a8366ad9da7a98876c63bd012","observation_id":"825f9578-8f73-459c-9c8b-16f2f5cc1acb","resolution":{"observed_at":"2026-08-15T23:50:44.081050Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2204.06105","last_updated":"2022-04-12T22:30:12Z","snapshot_observed_at":"2026-08-16T17:07:19.755382Z","submitted_at":"2022-04-12T22:30:12Z","title":"AGQA 2.0: An Updated Benchmark for Compositional Spatio-Temporal Reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2204.06105","snapshot_observed_at":"2026-08-15T23:50:43.919104Z","title":"Available: https://arxiv.org/abs/2204.06105","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.919104Z"},"links":{"cited_paper":"/paper/2204.06105","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:13ca045074bde37812c4baeea89b706b5b296b78f1153413373b231956295129","observation_id":"c137e4b7-3635-4d63-91ba-aceac1a5e0ed","resolution":{"observed_at":"2026-08-15T23:50:43.919104Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes"},"reference_resolution":{"displayed":60,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":36,"verified_exact":3,"verified_fuzzy":21},"total_outbound_references":60},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"thesis":"As of 17 August 2026, this Paper Citation Record lists 60 of 60 outbound references and 0 inbound Pith citation observations for arXiv:2505.03581."}