{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:3QLPTB3Y3SF3CEST4U5O6QFBU7","short_pith_number":"pith:3QLPTB3Y","schema_version":"1.0","canonical_sha256":"dc16f98778dc8bb11253e53aef40a1a7c656e94d89925fac7058dca0fabb5748","source":{"kind":"arxiv","id":"2303.14613","version":4},"attestation_state":"computed","paper":{"title":"GestureDiffuCLIP: Gesture Diffusion Model with CLIP Latents","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.GR"],"primary_cat":"cs.CV","authors_text":"Libin Liu, Tenglong Ao, Zeyi Zhang","submitted_at":"2023-03-26T03:35:46Z","abstract_excerpt":"The automatic generation of stylized co-speech gestures has recently received increasing attention. Previous systems typically allow style control via predefined text labels or example motion clips, which are often not flexible enough to convey user intent accurately. In this work, we present GestureDiffuCLIP, a neural network framework for synthesizing realistic, stylized co-speech gestures with flexible style control. We leverage the power of the large-scale Contrastive-Language-Image-Pre-training (CLIP) model and present a novel CLIP-guided mechanism that extracts efficient style representa"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2303.14613","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-03-26T03:35:46Z","cross_cats_sorted":["cs.GR"],"title_canon_sha256":"23abf7c2763d00f6b993aa32947c35bc071f3bb59dced33e510663fa9d143481","abstract_canon_sha256":"f9db115b1326fcb97e39d8b43178ec76e8dff76244eed256a43addef89d19e15"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:00:59.815053Z","signature_b64":"cGslazfXVX7piSYq4L/JNgoTQzkqegfW22R5idtfm/SrnRzjIhsU+Awj3rnt4c53Mls7PuZ1orGfphNZOoGzAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"dc16f98778dc8bb11253e53aef40a1a7c656e94d89925fac7058dca0fabb5748","last_reissued_at":"2026-07-05T07:00:59.814544Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:00:59.814544Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"GestureDiffuCLIP: Gesture Diffusion Model with CLIP Latents","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.GR"],"primary_cat":"cs.CV","authors_text":"Libin Liu, Tenglong Ao, Zeyi Zhang","submitted_at":"2023-03-26T03:35:46Z","abstract_excerpt":"The automatic generation of stylized co-speech gestures has recently received increasing attention. Previous systems typically allow style control via predefined text labels or example motion clips, which are often not flexible enough to convey user intent accurately. In this work, we present GestureDiffuCLIP, a neural network framework for synthesizing realistic, stylized co-speech gestures with flexible style control. We leverage the power of the large-scale Contrastive-Language-Image-Pre-training (CLIP) model and present a novel CLIP-guided mechanism that extracts efficient style representa"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2303.14613","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2303.14613/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2303.14613","created_at":"2026-07-05T07:00:59.814602+00:00"},{"alias_kind":"arxiv_version","alias_value":"2303.14613v4","created_at":"2026-07-05T07:00:59.814602+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2303.14613","created_at":"2026-07-05T07:00:59.814602+00:00"},{"alias_kind":"pith_short_12","alias_value":"3QLPTB3Y3SF3","created_at":"2026-07-05T07:00:59.814602+00:00"},{"alias_kind":"pith_short_16","alias_value":"3QLPTB3Y3SF3CEST","created_at":"2026-07-05T07:00:59.814602+00:00"},{"alias_kind":"pith_short_8","alias_value":"3QLPTB3Y","created_at":"2026-07-05T07:00:59.814602+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2412.20657","citing_title":"Diffgrasp: Whole-Body Grasping Synthesis Guided by Object Motion Using a Diffusion Model","ref_index":1,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3QLPTB3Y3SF3CEST4U5O6QFBU7","json":"https://pith.science/pith/3QLPTB3Y3SF3CEST4U5O6QFBU7.json","graph_json":"https://pith.science/api/pith-number/3QLPTB3Y3SF3CEST4U5O6QFBU7/graph.json","events_json":"https://pith.science/api/pith-number/3QLPTB3Y3SF3CEST4U5O6QFBU7/events.json","paper":"https://pith.science/paper/3QLPTB3Y"},"agent_actions":{"view_html":"https://pith.science/pith/3QLPTB3Y3SF3CEST4U5O6QFBU7","download_json":"https://pith.science/pith/3QLPTB3Y3SF3CEST4U5O6QFBU7.json","view_paper":"https://pith.science/paper/3QLPTB3Y","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2303.14613&json=true","fetch_graph":"https://pith.science/api/pith-number/3QLPTB3Y3SF3CEST4U5O6QFBU7/graph.json","fetch_events":"https://pith.science/api/pith-number/3QLPTB3Y3SF3CEST4U5O6QFBU7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3QLPTB3Y3SF3CEST4U5O6QFBU7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3QLPTB3Y3SF3CEST4U5O6QFBU7/action/storage_attestation","attest_author":"https://pith.science/pith/3QLPTB3Y3SF3CEST4U5O6QFBU7/action/author_attestation","sign_citation":"https://pith.science/pith/3QLPTB3Y3SF3CEST4U5O6QFBU7/action/citation_signature","submit_replication":"https://pith.science/pith/3QLPTB3Y3SF3CEST4U5O6QFBU7/action/replication_record"}},"created_at":"2026-07-05T07:00:59.814602+00:00","updated_at":"2026-07-05T07:00:59.814602+00:00"}