{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:E7AWZXVN5ABXQ32RSRLV5Z3AW4","short_pith_number":"pith:E7AWZXVN","schema_version":"1.0","canonical_sha256":"27c16cdeade803786f5194575ee760b72645d4a7bddbd08e292af67d314c53e8","source":{"kind":"arxiv","id":"2309.06680","version":3},"attestation_state":"computed","paper":{"title":"STUPD: A Synthetic Dataset for Spatial and Temporal Relation Reasoning","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Cheston Tan, Haidi Azaman, Palaash Agrawal","submitted_at":"2023-09-13T02:35:59Z","abstract_excerpt":"Understanding relations between objects is crucial for understanding the semantics of a visual scene. It is also an essential step in order to bridge visual and language models. However, current state-of-the-art computer vision models still lack the ability to perform spatial reasoning well. Existing datasets mostly cover a relatively small number of spatial relations, all of which are static relations that do not intrinsically involve motion. In this paper, we propose the Spatial and Temporal Understanding of Prepositions Dataset (STUPD) -- a large-scale video dataset for understanding static"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.06680","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2023-09-13T02:35:59Z","cross_cats_sorted":[],"title_canon_sha256":"412a143c35cd686d10d6f8ea7e0e0d966f503a206da044ef7bfc046ebdc5d90a","abstract_canon_sha256":"40137e4db26efc28f9d0c25ba785dfba7cbeb09f9d8e610af10dd26128aba427"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:21:02.146449Z","signature_b64":"WHMMTEardbcXQaAaeZ1JhJ/5cLyzXO+D7r4Ghxro1lS/JUcQQALmZhpCuYYfimHyrg5T8gUEMai6FXcTTZfMAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"27c16cdeade803786f5194575ee760b72645d4a7bddbd08e292af67d314c53e8","last_reissued_at":"2026-07-05T10:21:02.145971Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:21:02.145971Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"STUPD: A Synthetic Dataset for Spatial and Temporal Relation Reasoning","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Cheston Tan, Haidi Azaman, Palaash Agrawal","submitted_at":"2023-09-13T02:35:59Z","abstract_excerpt":"Understanding relations between objects is crucial for understanding the semantics of a visual scene. It is also an essential step in order to bridge visual and language models. However, current state-of-the-art computer vision models still lack the ability to perform spatial reasoning well. Existing datasets mostly cover a relatively small number of spatial relations, all of which are static relations that do not intrinsically involve motion. In this paper, we propose the Spatial and Temporal Understanding of Prepositions Dataset (STUPD) -- a large-scale video dataset for understanding static"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.06680","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.06680/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.06680","created_at":"2026-07-05T10:21:02.146022+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.06680v3","created_at":"2026-07-05T10:21:02.146022+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.06680","created_at":"2026-07-05T10:21:02.146022+00:00"},{"alias_kind":"pith_short_12","alias_value":"E7AWZXVN5ABX","created_at":"2026-07-05T10:21:02.146022+00:00"},{"alias_kind":"pith_short_16","alias_value":"E7AWZXVN5ABXQ32R","created_at":"2026-07-05T10:21:02.146022+00:00"},{"alias_kind":"pith_short_8","alias_value":"E7AWZXVN","created_at":"2026-07-05T10:21:02.146022+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2504.20648","citing_title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","ref_index":3,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/E7AWZXVN5ABXQ32RSRLV5Z3AW4","json":"https://pith.science/pith/E7AWZXVN5ABXQ32RSRLV5Z3AW4.json","graph_json":"https://pith.science/api/pith-number/E7AWZXVN5ABXQ32RSRLV5Z3AW4/graph.json","events_json":"https://pith.science/api/pith-number/E7AWZXVN5ABXQ32RSRLV5Z3AW4/events.json","paper":"https://pith.science/paper/E7AWZXVN"},"agent_actions":{"view_html":"https://pith.science/pith/E7AWZXVN5ABXQ32RSRLV5Z3AW4","download_json":"https://pith.science/pith/E7AWZXVN5ABXQ32RSRLV5Z3AW4.json","view_paper":"https://pith.science/paper/E7AWZXVN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.06680&json=true","fetch_graph":"https://pith.science/api/pith-number/E7AWZXVN5ABXQ32RSRLV5Z3AW4/graph.json","fetch_events":"https://pith.science/api/pith-number/E7AWZXVN5ABXQ32RSRLV5Z3AW4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/E7AWZXVN5ABXQ32RSRLV5Z3AW4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/E7AWZXVN5ABXQ32RSRLV5Z3AW4/action/storage_attestation","attest_author":"https://pith.science/pith/E7AWZXVN5ABXQ32RSRLV5Z3AW4/action/author_attestation","sign_citation":"https://pith.science/pith/E7AWZXVN5ABXQ32RSRLV5Z3AW4/action/citation_signature","submit_replication":"https://pith.science/pith/E7AWZXVN5ABXQ32RSRLV5Z3AW4/action/replication_record"}},"created_at":"2026-07-05T10:21:02.146022+00:00","updated_at":"2026-07-05T10:21:02.146022+00:00"}