{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:SBIA3PMLIPWV3G5ATJWEE2U56W","short_pith_number":"pith:SBIA3PML","schema_version":"1.0","canonical_sha256":"90500dbd8b43ed5d9ba09a6c426a9df59eb02865079eef138acdd5d94f424e12","source":{"kind":"arxiv","id":"2505.07615","version":2},"attestation_state":"computed","paper":{"title":"Diffused Responsibility: Analyzing the Energy Consumption of Generative Text-to-Audio Diffusion Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.SD"],"primary_cat":"eess.AS","authors_text":"Fabio Antonacci, Francesca Ronchini, Luca Comanducci, Riccardo Passoni, Romain Serizel","submitted_at":"2025-05-12T14:36:47Z","abstract_excerpt":"Text-to-audio models have recently emerged as a powerful technology for generating sound from textual descriptions. However, their high computational demands raise concerns about energy consumption and environmental impact. In this paper, we conduct an analysis of the energy usage of 7 state-of-the-art text-to-audio diffusion-based generative models, evaluating to what extent variations in generation parameters affect energy consumption at inference time. We also aim to identify an optimal balance between audio quality and energy consumption by considering Pareto-optimal solutions across all s"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.07615","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"eess.AS","submitted_at":"2025-05-12T14:36:47Z","cross_cats_sorted":["cs.AI","cs.LG","cs.SD"],"title_canon_sha256":"06d71bc6e4e235cc4a2904fa0a667b284dcbc81b217cf2cd59b172c403a08ecf","abstract_canon_sha256":"e21a36eaa305289d9b6d8484ac514e2b70d78d146fc843ec7a8b8e9e7861a14a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:38:07.118443Z","signature_b64":"m28gVdiBo0otF8G9VGhLXLRP7rwklDXj5/XdKO2yUCFq3y7tZrzDAZZBOvY07xQCZeBYPNvIT6+TWiHlegyaBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"90500dbd8b43ed5d9ba09a6c426a9df59eb02865079eef138acdd5d94f424e12","last_reissued_at":"2026-07-05T11:38:07.117886Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:38:07.117886Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Diffused Responsibility: Analyzing the Energy Consumption of Generative Text-to-Audio Diffusion Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.SD"],"primary_cat":"eess.AS","authors_text":"Fabio Antonacci, Francesca Ronchini, Luca Comanducci, Riccardo Passoni, Romain Serizel","submitted_at":"2025-05-12T14:36:47Z","abstract_excerpt":"Text-to-audio models have recently emerged as a powerful technology for generating sound from textual descriptions. However, their high computational demands raise concerns about energy consumption and environmental impact. In this paper, we conduct an analysis of the energy usage of 7 state-of-the-art text-to-audio diffusion-based generative models, evaluating to what extent variations in generation parameters affect energy consumption at inference time. We also aim to identify an optimal balance between audio quality and energy consumption by considering Pareto-optimal solutions across all s"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.07615","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.07615/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.07615","created_at":"2026-07-05T11:38:07.117948+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.07615v2","created_at":"2026-07-05T11:38:07.117948+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.07615","created_at":"2026-07-05T11:38:07.117948+00:00"},{"alias_kind":"pith_short_12","alias_value":"SBIA3PMLIPWV","created_at":"2026-07-05T11:38:07.117948+00:00"},{"alias_kind":"pith_short_16","alias_value":"SBIA3PMLIPWV3G5A","created_at":"2026-07-05T11:38:07.117948+00:00"},{"alias_kind":"pith_short_8","alias_value":"SBIA3PML","created_at":"2026-07-05T11:38:07.117948+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.00329","citing_title":"Fast Text-to-Audio Generation with One-Step Sampling via Energy-Scoring and Auxiliary Contextual Representation Distillation","ref_index":22,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SBIA3PMLIPWV3G5ATJWEE2U56W","json":"https://pith.science/pith/SBIA3PMLIPWV3G5ATJWEE2U56W.json","graph_json":"https://pith.science/api/pith-number/SBIA3PMLIPWV3G5ATJWEE2U56W/graph.json","events_json":"https://pith.science/api/pith-number/SBIA3PMLIPWV3G5ATJWEE2U56W/events.json","paper":"https://pith.science/paper/SBIA3PML"},"agent_actions":{"view_html":"https://pith.science/pith/SBIA3PMLIPWV3G5ATJWEE2U56W","download_json":"https://pith.science/pith/SBIA3PMLIPWV3G5ATJWEE2U56W.json","view_paper":"https://pith.science/paper/SBIA3PML","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.07615&json=true","fetch_graph":"https://pith.science/api/pith-number/SBIA3PMLIPWV3G5ATJWEE2U56W/graph.json","fetch_events":"https://pith.science/api/pith-number/SBIA3PMLIPWV3G5ATJWEE2U56W/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SBIA3PMLIPWV3G5ATJWEE2U56W/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SBIA3PMLIPWV3G5ATJWEE2U56W/action/storage_attestation","attest_author":"https://pith.science/pith/SBIA3PMLIPWV3G5ATJWEE2U56W/action/author_attestation","sign_citation":"https://pith.science/pith/SBIA3PMLIPWV3G5ATJWEE2U56W/action/citation_signature","submit_replication":"https://pith.science/pith/SBIA3PMLIPWV3G5ATJWEE2U56W/action/replication_record"}},"created_at":"2026-07-05T11:38:07.117948+00:00","updated_at":"2026-07-05T11:38:07.117948+00:00"}