{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:DNI5FHUAFZGTKOPYUN4S6WDFIA","short_pith_number":"pith:DNI5FHUA","schema_version":"1.0","canonical_sha256":"1b51d29e802e4d3539f8a3792f5865403319fe347cd7f0c84e140ef81453a5f8","source":{"kind":"arxiv","id":"2505.11326","version":1},"attestation_state":"computed","paper":{"title":"Temporally-Grounded Language Generation: A Benchmark for Real-Time Vision-Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Joyce Chai, Keunwoo Peter Yu","submitted_at":"2025-05-16T14:48:30Z","abstract_excerpt":"Vision-language models (VLMs) have shown remarkable progress in offline tasks such as image captioning and video question answering. However, real-time interactive environments impose new demands on VLMs, requiring them to generate utterances that are not only semantically accurate but also precisely timed. We identify two core capabilities necessary for such settings -- $\\textit{perceptual updating}$ and $\\textit{contingency awareness}$ -- and propose a new benchmark task, $\\textbf{Temporally-Grounded Language Generation (TGLG)}$, to evaluate them. TGLG requires models to generate utterances "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.11326","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-05-16T14:48:30Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"8922981d2b8791488a954d103242d0b8df85c41c5a2e2b757be79f32791538b0","abstract_canon_sha256":"e22f77aa55ae82eff089b65cc890bc55335e59c338bd53622cf43d4c7c8ce79d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:04:14.387915Z","signature_b64":"ZYYzwONS2qy8ZEakcJi4/ADgyfUiXSfKtqSaHqwI9HCO+EqjtqiERudenVbeHeSZ22YUQvblIfngkELVOSBZAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1b51d29e802e4d3539f8a3792f5865403319fe347cd7f0c84e140ef81453a5f8","last_reissued_at":"2026-07-05T11:04:14.387430Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:04:14.387430Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Temporally-Grounded Language Generation: A Benchmark for Real-Time Vision-Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Joyce Chai, Keunwoo Peter Yu","submitted_at":"2025-05-16T14:48:30Z","abstract_excerpt":"Vision-language models (VLMs) have shown remarkable progress in offline tasks such as image captioning and video question answering. However, real-time interactive environments impose new demands on VLMs, requiring them to generate utterances that are not only semantically accurate but also precisely timed. We identify two core capabilities necessary for such settings -- $\\textit{perceptual updating}$ and $\\textit{contingency awareness}$ -- and propose a new benchmark task, $\\textbf{Temporally-Grounded Language Generation (TGLG)}$, to evaluate them. TGLG requires models to generate utterances "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.11326","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.11326/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.11326","created_at":"2026-07-05T11:04:14.387489+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.11326v1","created_at":"2026-07-05T11:04:14.387489+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.11326","created_at":"2026-07-05T11:04:14.387489+00:00"},{"alias_kind":"pith_short_12","alias_value":"DNI5FHUAFZGT","created_at":"2026-07-05T11:04:14.387489+00:00"},{"alias_kind":"pith_short_16","alias_value":"DNI5FHUAFZGTKOPY","created_at":"2026-07-05T11:04:14.387489+00:00"},{"alias_kind":"pith_short_8","alias_value":"DNI5FHUA","created_at":"2026-07-05T11:04:14.387489+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DNI5FHUAFZGTKOPYUN4S6WDFIA","json":"https://pith.science/pith/DNI5FHUAFZGTKOPYUN4S6WDFIA.json","graph_json":"https://pith.science/api/pith-number/DNI5FHUAFZGTKOPYUN4S6WDFIA/graph.json","events_json":"https://pith.science/api/pith-number/DNI5FHUAFZGTKOPYUN4S6WDFIA/events.json","paper":"https://pith.science/paper/DNI5FHUA"},"agent_actions":{"view_html":"https://pith.science/pith/DNI5FHUAFZGTKOPYUN4S6WDFIA","download_json":"https://pith.science/pith/DNI5FHUAFZGTKOPYUN4S6WDFIA.json","view_paper":"https://pith.science/paper/DNI5FHUA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.11326&json=true","fetch_graph":"https://pith.science/api/pith-number/DNI5FHUAFZGTKOPYUN4S6WDFIA/graph.json","fetch_events":"https://pith.science/api/pith-number/DNI5FHUAFZGTKOPYUN4S6WDFIA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DNI5FHUAFZGTKOPYUN4S6WDFIA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DNI5FHUAFZGTKOPYUN4S6WDFIA/action/storage_attestation","attest_author":"https://pith.science/pith/DNI5FHUAFZGTKOPYUN4S6WDFIA/action/author_attestation","sign_citation":"https://pith.science/pith/DNI5FHUAFZGTKOPYUN4S6WDFIA/action/citation_signature","submit_replication":"https://pith.science/pith/DNI5FHUAFZGTKOPYUN4S6WDFIA/action/replication_record"}},"created_at":"2026-07-05T11:04:14.387489+00:00","updated_at":"2026-07-05T11:04:14.387489+00:00"}