{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:ABQPKA5UHO6TPOLKGMSPLNUWFL","short_pith_number":"pith:ABQPKA5U","schema_version":"1.0","canonical_sha256":"0060f503b43bbd37b96a3324f5b6962acbc924461a4eba50d3aa0598cc8ce4c6","source":{"kind":"arxiv","id":"2311.09861","version":4},"attestation_state":"computed","paper":{"title":"ConceptPsy:A Benchmark Suite with Conceptual Comprehensiveness in Psychology","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Anqi Li, Hongliang He, Huachuan Qiu, Junlei Zhang, Lizhi Ma, Nirui Song, Shuai Zhang, Shuyuan He, Yong Dai, Zhanchao Zhou, Zhenzhong Lan","submitted_at":"2023-11-16T12:43:18Z","abstract_excerpt":"The critical field of psychology necessitates a comprehensive benchmark to enhance the evaluation and development of domain-specific Large Language Models (LLMs). Existing MMLU-type benchmarks, such as C-EVAL and CMMLU, include psychology-related subjects, but their limited number of questions and lack of systematic concept sampling strategies mean they cannot cover the concepts required in psychology. Consequently, despite their broad subject coverage, these benchmarks lack the necessary depth in the psychology domain, making them inadequate as psychology-specific evaluation suite. To address"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.09861","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-11-16T12:43:18Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"a55a6c7773976f90391b2322feb00e558c1a872e5070327f5c56e5074fd09956","abstract_canon_sha256":"ec97bd1c1164c505b607f7ad22d5a0aa411a4cc865aa4cbd11406a5dcf3db1ae"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:32:33.006645Z","signature_b64":"QjRh3JIpQ9gvxTj9k1Qkjb7DrUE1HYLp0obY8DsYqJENIOYe6orWbCuSK+mQEqPxxkpiDkbpuCzSYfKZf8H6CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0060f503b43bbd37b96a3324f5b6962acbc924461a4eba50d3aa0598cc8ce4c6","last_reissued_at":"2026-07-05T08:32:33.006156Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:32:33.006156Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ConceptPsy:A Benchmark Suite with Conceptual Comprehensiveness in Psychology","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Anqi Li, Hongliang He, Huachuan Qiu, Junlei Zhang, Lizhi Ma, Nirui Song, Shuai Zhang, Shuyuan He, Yong Dai, Zhanchao Zhou, Zhenzhong Lan","submitted_at":"2023-11-16T12:43:18Z","abstract_excerpt":"The critical field of psychology necessitates a comprehensive benchmark to enhance the evaluation and development of domain-specific Large Language Models (LLMs). Existing MMLU-type benchmarks, such as C-EVAL and CMMLU, include psychology-related subjects, but their limited number of questions and lack of systematic concept sampling strategies mean they cannot cover the concepts required in psychology. Consequently, despite their broad subject coverage, these benchmarks lack the necessary depth in the psychology domain, making them inadequate as psychology-specific evaluation suite. To address"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.09861","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.09861/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.09861","created_at":"2026-07-05T08:32:33.006211+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.09861v4","created_at":"2026-07-05T08:32:33.006211+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.09861","created_at":"2026-07-05T08:32:33.006211+00:00"},{"alias_kind":"pith_short_12","alias_value":"ABQPKA5UHO6T","created_at":"2026-07-05T08:32:33.006211+00:00"},{"alias_kind":"pith_short_16","alias_value":"ABQPKA5UHO6TPOLK","created_at":"2026-07-05T08:32:33.006211+00:00"},{"alias_kind":"pith_short_8","alias_value":"ABQPKA5U","created_at":"2026-07-05T08:32:33.006211+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.09825","citing_title":"KRISTEVA: Close Reading as a Novel Task for Benchmarking Interpretive Reasoning","ref_index":14,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ABQPKA5UHO6TPOLKGMSPLNUWFL","json":"https://pith.science/pith/ABQPKA5UHO6TPOLKGMSPLNUWFL.json","graph_json":"https://pith.science/api/pith-number/ABQPKA5UHO6TPOLKGMSPLNUWFL/graph.json","events_json":"https://pith.science/api/pith-number/ABQPKA5UHO6TPOLKGMSPLNUWFL/events.json","paper":"https://pith.science/paper/ABQPKA5U"},"agent_actions":{"view_html":"https://pith.science/pith/ABQPKA5UHO6TPOLKGMSPLNUWFL","download_json":"https://pith.science/pith/ABQPKA5UHO6TPOLKGMSPLNUWFL.json","view_paper":"https://pith.science/paper/ABQPKA5U","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.09861&json=true","fetch_graph":"https://pith.science/api/pith-number/ABQPKA5UHO6TPOLKGMSPLNUWFL/graph.json","fetch_events":"https://pith.science/api/pith-number/ABQPKA5UHO6TPOLKGMSPLNUWFL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ABQPKA5UHO6TPOLKGMSPLNUWFL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ABQPKA5UHO6TPOLKGMSPLNUWFL/action/storage_attestation","attest_author":"https://pith.science/pith/ABQPKA5UHO6TPOLKGMSPLNUWFL/action/author_attestation","sign_citation":"https://pith.science/pith/ABQPKA5UHO6TPOLKGMSPLNUWFL/action/citation_signature","submit_replication":"https://pith.science/pith/ABQPKA5UHO6TPOLKGMSPLNUWFL/action/replication_record"}},"created_at":"2026-07-05T08:32:33.006211+00:00","updated_at":"2026-07-05T08:32:33.006211+00:00"}