{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:XW34DAD6CFKBVOJPZYB22G625O","short_pith_number":"pith:XW34DAD6","schema_version":"1.0","canonical_sha256":"bdb7c1807e11541ab92fce03ad1bdaeb9ccc8412100eb9f384ba5a21f8f56266","source":{"kind":"arxiv","id":"2409.13903","version":1},"attestation_state":"computed","paper":{"title":"CI-Bench: Benchmarking Contextual Integrity of AI Assistants on Synthetic Data","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Borja Balle, Diane Wan, Eugene Bagdasarian, Matthew Abueg, Ren Yi, Sahra Ghalebikesabi, Shawn O'Banion, Stefan Mellem, Zhao Cheng","submitted_at":"2024-09-20T21:14:36Z","abstract_excerpt":"Advances in generative AI point towards a new era of personalized applications that perform diverse tasks on behalf of users. While general AI assistants have yet to fully emerge, their potential to share personal data raises significant privacy challenges. This paper introduces CI-Bench, a comprehensive synthetic benchmark for evaluating the ability of AI assistants to protect personal information during model inference. Leveraging the Contextual Integrity framework, our benchmark enables systematic assessment of information flow across important context dimensions, including roles, informati"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.13903","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.AI","submitted_at":"2024-09-20T21:14:36Z","cross_cats_sorted":[],"title_canon_sha256":"1770d5fe2ae69feb0e04d58828a0f0c4beec1c0ab0580f6cbe9d84d0ca37bb75","abstract_canon_sha256":"6aff142051549619939202a0a40e3989f8036a31cc89d84556fcfbc1a8cfcad0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:10:08.547798Z","signature_b64":"i9UkoTXvx8/pKXLsoPaGaRGXnZWiUXRQ36n5jTN7Hgoc3tRIYmUnOsRv5WaRu0ovlDHvD1eS2AvmXHySL86mCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bdb7c1807e11541ab92fce03ad1bdaeb9ccc8412100eb9f384ba5a21f8f56266","last_reissued_at":"2026-07-05T09:10:08.547366Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:10:08.547366Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CI-Bench: Benchmarking Contextual Integrity of AI Assistants on Synthetic Data","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Borja Balle, Diane Wan, Eugene Bagdasarian, Matthew Abueg, Ren Yi, Sahra Ghalebikesabi, Shawn O'Banion, Stefan Mellem, Zhao Cheng","submitted_at":"2024-09-20T21:14:36Z","abstract_excerpt":"Advances in generative AI point towards a new era of personalized applications that perform diverse tasks on behalf of users. While general AI assistants have yet to fully emerge, their potential to share personal data raises significant privacy challenges. This paper introduces CI-Bench, a comprehensive synthetic benchmark for evaluating the ability of AI assistants to protect personal information during model inference. Leveraging the Contextual Integrity framework, our benchmark enables systematic assessment of information flow across important context dimensions, including roles, informati"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.13903","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.13903/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.13903","created_at":"2026-07-05T09:10:08.547424+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.13903v1","created_at":"2026-07-05T09:10:08.547424+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.13903","created_at":"2026-07-05T09:10:08.547424+00:00"},{"alias_kind":"pith_short_12","alias_value":"XW34DAD6CFKB","created_at":"2026-07-05T09:10:08.547424+00:00"},{"alias_kind":"pith_short_16","alias_value":"XW34DAD6CFKBVOJP","created_at":"2026-07-05T09:10:08.547424+00:00"},{"alias_kind":"pith_short_8","alias_value":"XW34DAD6","created_at":"2026-07-05T09:10:08.547424+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05318","citing_title":"PiSAs: Benchmarking Contextual Integrity in Multi-User Agentic Systems","ref_index":4,"is_internal_anchor":true},{"citing_arxiv_id":"2606.26627","citing_title":"Agents That Know Too Much: A Data-Centric Survey of Privacy in LLM Agents","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2606.23217","citing_title":"MuPPET: A Benchmark for Contextual Privacy of LLM Assistants in Multi-Party Conversations","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04067","citing_title":"Need to Know: Contextual-Integrity-Grounded Query Rewriting for Privacy-Conscious LLM Delegation","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23989","citing_title":"Towards trustworthy agentic AI: a comprehensive survey of safety, robustness, privacy, and system security","ref_index":133,"is_internal_anchor":false},{"citing_arxiv_id":"2505.14549","citing_title":"Can Large Language Models Really Recognize Your Name?","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20258","citing_title":"It Takes Two: Complementary Self-Distillation for Contextual Integrity in LLMs","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16630","citing_title":"PrivScope: Task-scoped Disclosure Control for Hybrid Agentic Systems","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17830","citing_title":"Remembering More, Risking More: Longitudinal Safety Risks in Memory-Equipped LLM Agents","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08378","citing_title":"Reinforcement Learning for Scalable and Trustworthy Intelligent Systems","ref_index":130,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21308","citing_title":"CI-Work: Benchmarking Contextual Integrity in Enterprise LLM Agents","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12308","citing_title":"ContextLens: Modeling Imperfect Privacy and Safety Context for Legal Compliance","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XW34DAD6CFKBVOJPZYB22G625O","json":"https://pith.science/pith/XW34DAD6CFKBVOJPZYB22G625O.json","graph_json":"https://pith.science/api/pith-number/XW34DAD6CFKBVOJPZYB22G625O/graph.json","events_json":"https://pith.science/api/pith-number/XW34DAD6CFKBVOJPZYB22G625O/events.json","paper":"https://pith.science/paper/XW34DAD6"},"agent_actions":{"view_html":"https://pith.science/pith/XW34DAD6CFKBVOJPZYB22G625O","download_json":"https://pith.science/pith/XW34DAD6CFKBVOJPZYB22G625O.json","view_paper":"https://pith.science/paper/XW34DAD6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.13903&json=true","fetch_graph":"https://pith.science/api/pith-number/XW34DAD6CFKBVOJPZYB22G625O/graph.json","fetch_events":"https://pith.science/api/pith-number/XW34DAD6CFKBVOJPZYB22G625O/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XW34DAD6CFKBVOJPZYB22G625O/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XW34DAD6CFKBVOJPZYB22G625O/action/storage_attestation","attest_author":"https://pith.science/pith/XW34DAD6CFKBVOJPZYB22G625O/action/author_attestation","sign_citation":"https://pith.science/pith/XW34DAD6CFKBVOJPZYB22G625O/action/citation_signature","submit_replication":"https://pith.science/pith/XW34DAD6CFKBVOJPZYB22G625O/action/replication_record"}},"created_at":"2026-07-05T09:10:08.547424+00:00","updated_at":"2026-07-05T09:10:08.547424+00:00"}