{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:EJSKY2OZG3LE4MVJMMHMCZAQL2","short_pith_number":"pith:EJSKY2OZ","schema_version":"1.0","canonical_sha256":"2264ac69d936d64e32a9630ec164105ea62486f4740afd9c05741659e20626cc","source":{"kind":"arxiv","id":"2602.07267","version":2},"attestation_state":"computed","paper":{"title":"BRIDGE: Predicting Human Task Completion Time From Model Performance","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.AI","authors_text":"Dzmitry Bahdanau, Fengyuan Liu, Hugo Larochelle, Jay Gala, Nilaksh, Siva Reddy","submitted_at":"2026-02-06T23:36:11Z","abstract_excerpt":"Evaluating the real-world capabilities of AI systems requires grounding benchmark performance in human-interpretable measures of task difficulty. Existing approaches that rely on direct human task completion time annotations are costly, noisy, and difficult to scale across benchmarks. In this work, we propose BRIDGE, a unified psychometric framework that learns a latent difficulty scale from model responses and anchors it to human task completion time. Using a two-parameter logistic Item Response Theory model, we jointly estimate latent task difficulty and model capability from model performan"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2602.07267","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-02-06T23:36:11Z","cross_cats_sorted":["cs.CL","cs.LG"],"title_canon_sha256":"8869599c7cb010b9f80f5216dc8f2d0b5cc9c9cc687d5103d3aee9f6b2eb1c55","abstract_canon_sha256":"83b0401ecf5ceeae3ae87990a512f5b62aaeab2f185f9505b3f08474da2cdf3b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-03T01:17:17.326794Z","signature_b64":"QRBRvLX99FmNc8ctc0x2Y4lWPGy6VolYAoegCR4rBbXCDs/HHREbfbAiopWnfGjxPG0i4WmttKYIlNEHieysAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2264ac69d936d64e32a9630ec164105ea62486f4740afd9c05741659e20626cc","last_reissued_at":"2026-07-03T01:17:17.326332Z","signature_status":"signed_v1","first_computed_at":"2026-07-03T01:17:17.326332Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"BRIDGE: Predicting Human Task Completion Time From Model Performance","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.AI","authors_text":"Dzmitry Bahdanau, Fengyuan Liu, Hugo Larochelle, Jay Gala, Nilaksh, Siva Reddy","submitted_at":"2026-02-06T23:36:11Z","abstract_excerpt":"Evaluating the real-world capabilities of AI systems requires grounding benchmark performance in human-interpretable measures of task difficulty. Existing approaches that rely on direct human task completion time annotations are costly, noisy, and difficult to scale across benchmarks. In this work, we propose BRIDGE, a unified psychometric framework that learns a latent difficulty scale from model responses and anchors it to human task completion time. Using a two-parameter logistic Item Response Theory model, we jointly estimate latent task difficulty and model capability from model performan"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2602.07267","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2602.07267/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2602.07267","created_at":"2026-07-03T01:17:17.326392+00:00"},{"alias_kind":"arxiv_version","alias_value":"2602.07267v2","created_at":"2026-07-03T01:17:17.326392+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2602.07267","created_at":"2026-07-03T01:17:17.326392+00:00"},{"alias_kind":"pith_short_12","alias_value":"EJSKY2OZG3LE","created_at":"2026-07-03T01:17:17.326392+00:00"},{"alias_kind":"pith_short_16","alias_value":"EJSKY2OZG3LE4MVJ","created_at":"2026-07-03T01:17:17.326392+00:00"},{"alias_kind":"pith_short_8","alias_value":"EJSKY2OZ","created_at":"2026-07-03T01:17:17.326392+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2608.05797","citing_title":"Predicting Task Difficulty Without Rollouts","ref_index":15,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EJSKY2OZG3LE4MVJMMHMCZAQL2","json":"https://pith.science/pith/EJSKY2OZG3LE4MVJMMHMCZAQL2.json","graph_json":"https://pith.science/api/pith-number/EJSKY2OZG3LE4MVJMMHMCZAQL2/graph.json","events_json":"https://pith.science/api/pith-number/EJSKY2OZG3LE4MVJMMHMCZAQL2/events.json","paper":"https://pith.science/paper/EJSKY2OZ"},"agent_actions":{"view_html":"https://pith.science/pith/EJSKY2OZG3LE4MVJMMHMCZAQL2","download_json":"https://pith.science/pith/EJSKY2OZG3LE4MVJMMHMCZAQL2.json","view_paper":"https://pith.science/paper/EJSKY2OZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2602.07267&json=true","fetch_graph":"https://pith.science/api/pith-number/EJSKY2OZG3LE4MVJMMHMCZAQL2/graph.json","fetch_events":"https://pith.science/api/pith-number/EJSKY2OZG3LE4MVJMMHMCZAQL2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EJSKY2OZG3LE4MVJMMHMCZAQL2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EJSKY2OZG3LE4MVJMMHMCZAQL2/action/storage_attestation","attest_author":"https://pith.science/pith/EJSKY2OZG3LE4MVJMMHMCZAQL2/action/author_attestation","sign_citation":"https://pith.science/pith/EJSKY2OZG3LE4MVJMMHMCZAQL2/action/citation_signature","submit_replication":"https://pith.science/pith/EJSKY2OZG3LE4MVJMMHMCZAQL2/action/replication_record"}},"created_at":"2026-07-03T01:17:17.326392+00:00","updated_at":"2026-07-03T01:17:17.326392+00:00"}