{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:R5MJBHRM5DSGDXCIUZJLR5ADZV","short_pith_number":"pith:R5MJBHRM","schema_version":"1.0","canonical_sha256":"8f58909e2ce8e461dc48a652b8f403cd7087e5162e1c6dafbff7effb3da9129d","source":{"kind":"arxiv","id":"2110.05287","version":1},"attestation_state":"computed","paper":{"title":"TEET! Tunisian Dataset for Toxic Speech Detection","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Hatem Haddad, Heger Arfaoui, Mayssa Kchaou, Slim Gharbi","submitted_at":"2021-10-11T14:00:08Z","abstract_excerpt":"The complete freedom of expression in social media has its costs especially in spreading harmful and abusive content that may induce people to act accordingly. Therefore, the need of detecting automatically such a content becomes an urgent task that will help and enhance the efficiency in limiting this toxic spread. Compared to other Arabic dialects which are mostly based on MSA, the Tunisian dialect is a combination of many other languages like MSA, Tamazight, Italian and French. Because of its rich language, dealing with NLP problems can be challenging due to the lack of large annotated data"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2110.05287","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/publicdomain/zero/1.0/","primary_cat":"cs.CL","submitted_at":"2021-10-11T14:00:08Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"02f990f73def662f1951024ae99aa8404cf7be33134f9ba992a132aff253c857","abstract_canon_sha256":"3ceb5c57b7b151aa385cbb8c70fdbfeb610309940a0ce216ad900deff4479b0c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:21:27.452800Z","signature_b64":"LgQMvQ3lg7JQSQev0qERyWKT+kU4CYLm3oGsIXVRSt8kzglvIJVFR8XOyht9RPN8hFK4hDHwbgENYJ4gv5Z1Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8f58909e2ce8e461dc48a652b8f403cd7087e5162e1c6dafbff7effb3da9129d","last_reissued_at":"2026-07-05T03:21:27.452433Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:21:27.452433Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TEET! Tunisian Dataset for Toxic Speech Detection","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Hatem Haddad, Heger Arfaoui, Mayssa Kchaou, Slim Gharbi","submitted_at":"2021-10-11T14:00:08Z","abstract_excerpt":"The complete freedom of expression in social media has its costs especially in spreading harmful and abusive content that may induce people to act accordingly. Therefore, the need of detecting automatically such a content becomes an urgent task that will help and enhance the efficiency in limiting this toxic spread. Compared to other Arabic dialects which are mostly based on MSA, the Tunisian dialect is a combination of many other languages like MSA, Tamazight, Italian and French. Because of its rich language, dealing with NLP problems can be challenging due to the lack of large annotated data"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2110.05287","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2110.05287/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2110.05287","created_at":"2026-07-05T03:21:27.452496+00:00"},{"alias_kind":"arxiv_version","alias_value":"2110.05287v1","created_at":"2026-07-05T03:21:27.452496+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2110.05287","created_at":"2026-07-05T03:21:27.452496+00:00"},{"alias_kind":"pith_short_12","alias_value":"R5MJBHRM5DSG","created_at":"2026-07-05T03:21:27.452496+00:00"},{"alias_kind":"pith_short_16","alias_value":"R5MJBHRM5DSGDXCI","created_at":"2026-07-05T03:21:27.452496+00:00"},{"alias_kind":"pith_short_8","alias_value":"R5MJBHRM","created_at":"2026-07-05T03:21:27.452496+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/R5MJBHRM5DSGDXCIUZJLR5ADZV","json":"https://pith.science/pith/R5MJBHRM5DSGDXCIUZJLR5ADZV.json","graph_json":"https://pith.science/api/pith-number/R5MJBHRM5DSGDXCIUZJLR5ADZV/graph.json","events_json":"https://pith.science/api/pith-number/R5MJBHRM5DSGDXCIUZJLR5ADZV/events.json","paper":"https://pith.science/paper/R5MJBHRM"},"agent_actions":{"view_html":"https://pith.science/pith/R5MJBHRM5DSGDXCIUZJLR5ADZV","download_json":"https://pith.science/pith/R5MJBHRM5DSGDXCIUZJLR5ADZV.json","view_paper":"https://pith.science/paper/R5MJBHRM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2110.05287&json=true","fetch_graph":"https://pith.science/api/pith-number/R5MJBHRM5DSGDXCIUZJLR5ADZV/graph.json","fetch_events":"https://pith.science/api/pith-number/R5MJBHRM5DSGDXCIUZJLR5ADZV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/R5MJBHRM5DSGDXCIUZJLR5ADZV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/R5MJBHRM5DSGDXCIUZJLR5ADZV/action/storage_attestation","attest_author":"https://pith.science/pith/R5MJBHRM5DSGDXCIUZJLR5ADZV/action/author_attestation","sign_citation":"https://pith.science/pith/R5MJBHRM5DSGDXCIUZJLR5ADZV/action/citation_signature","submit_replication":"https://pith.science/pith/R5MJBHRM5DSGDXCIUZJLR5ADZV/action/replication_record"}},"created_at":"2026-07-05T03:21:27.452496+00:00","updated_at":"2026-07-05T03:21:27.452496+00:00"}