{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:IQVKRBWJAWR6S2GPI6ZQLNAGZZ","short_pith_number":"pith:IQVKRBWJ","schema_version":"1.0","canonical_sha256":"442aa886c905a3e968cf47b305b406ce4de8eba8c939d682419ae37886280896","source":{"kind":"arxiv","id":"2311.18812","version":1},"attestation_state":"computed","paper":{"title":"What Do Llamas Really Think? Revealing Preference Biases in Language Model Representations","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Ferhan Ture, Jimmy Lin, Raphael Tang, Xinyu Zhang","submitted_at":"2023-11-30T18:53:13Z","abstract_excerpt":"Do large language models (LLMs) exhibit sociodemographic biases, even when they decline to respond? To bypass their refusal to \"speak,\" we study this research question by probing contextualized embeddings and exploring whether this bias is encoded in its latent representations. We propose a logistic Bradley-Terry probe which predicts word pair preferences of LLMs from the words' hidden vectors. We first validate our probe on three pair preference tasks and thirteen LLMs, where we outperform the word embedding association test (WEAT), a standard approach in testing for implicit association, by "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.18812","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-11-30T18:53:13Z","cross_cats_sorted":[],"title_canon_sha256":"fec66557b5396d6bcaefddeb648e781f9447bfa012a458c0fed311e3d126ecad","abstract_canon_sha256":"61aafdae8d69f44bdce80197cf59207586a4bb42aeef1cae3bfbcd83e42ba27c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:18:46.942003Z","signature_b64":"aFZMwHsFGujjqY3EsCUehs/SnQlP5eJKK9dP7mq+F/LOXCQ8TpMDbiLPBJIidFtdeosY6N1sLirzJcp/nD3DCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"442aa886c905a3e968cf47b305b406ce4de8eba8c939d682419ae37886280896","last_reissued_at":"2026-07-05T07:18:46.941528Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:18:46.941528Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"What Do Llamas Really Think? Revealing Preference Biases in Language Model Representations","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Ferhan Ture, Jimmy Lin, Raphael Tang, Xinyu Zhang","submitted_at":"2023-11-30T18:53:13Z","abstract_excerpt":"Do large language models (LLMs) exhibit sociodemographic biases, even when they decline to respond? To bypass their refusal to \"speak,\" we study this research question by probing contextualized embeddings and exploring whether this bias is encoded in its latent representations. We propose a logistic Bradley-Terry probe which predicts word pair preferences of LLMs from the words' hidden vectors. We first validate our probe on three pair preference tasks and thirteen LLMs, where we outperform the word embedding association test (WEAT), a standard approach in testing for implicit association, by "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.18812","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.18812/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.18812","created_at":"2026-07-05T07:18:46.941589+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.18812v1","created_at":"2026-07-05T07:18:46.941589+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.18812","created_at":"2026-07-05T07:18:46.941589+00:00"},{"alias_kind":"pith_short_12","alias_value":"IQVKRBWJAWR6","created_at":"2026-07-05T07:18:46.941589+00:00"},{"alias_kind":"pith_short_16","alias_value":"IQVKRBWJAWR6S2GP","created_at":"2026-07-05T07:18:46.941589+00:00"},{"alias_kind":"pith_short_8","alias_value":"IQVKRBWJ","created_at":"2026-07-05T07:18:46.941589+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.22749","citing_title":"Representational Harms in LLM-Generated Narratives Against Global Majority Nationalities","ref_index":73,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IQVKRBWJAWR6S2GPI6ZQLNAGZZ","json":"https://pith.science/pith/IQVKRBWJAWR6S2GPI6ZQLNAGZZ.json","graph_json":"https://pith.science/api/pith-number/IQVKRBWJAWR6S2GPI6ZQLNAGZZ/graph.json","events_json":"https://pith.science/api/pith-number/IQVKRBWJAWR6S2GPI6ZQLNAGZZ/events.json","paper":"https://pith.science/paper/IQVKRBWJ"},"agent_actions":{"view_html":"https://pith.science/pith/IQVKRBWJAWR6S2GPI6ZQLNAGZZ","download_json":"https://pith.science/pith/IQVKRBWJAWR6S2GPI6ZQLNAGZZ.json","view_paper":"https://pith.science/paper/IQVKRBWJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.18812&json=true","fetch_graph":"https://pith.science/api/pith-number/IQVKRBWJAWR6S2GPI6ZQLNAGZZ/graph.json","fetch_events":"https://pith.science/api/pith-number/IQVKRBWJAWR6S2GPI6ZQLNAGZZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IQVKRBWJAWR6S2GPI6ZQLNAGZZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IQVKRBWJAWR6S2GPI6ZQLNAGZZ/action/storage_attestation","attest_author":"https://pith.science/pith/IQVKRBWJAWR6S2GPI6ZQLNAGZZ/action/author_attestation","sign_citation":"https://pith.science/pith/IQVKRBWJAWR6S2GPI6ZQLNAGZZ/action/citation_signature","submit_replication":"https://pith.science/pith/IQVKRBWJAWR6S2GPI6ZQLNAGZZ/action/replication_record"}},"created_at":"2026-07-05T07:18:46.941589+00:00","updated_at":"2026-07-05T07:18:46.941589+00:00"}