{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:HGXTSAEZPVFET4WW7UAXZUQNUM","short_pith_number":"pith:HGXTSAEZ","schema_version":"1.0","canonical_sha256":"39af3900997d4a49f2d6fd017cd20da302b453542dff5d2831d3fbea3ecd4d12","source":{"kind":"arxiv","id":"2406.17158","version":1},"attestation_state":"computed","paper":{"title":"DEXTER: A Benchmark for open-domain Complex Question Answering using LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.IR"],"primary_cat":"cs.CL","authors_text":"Avishek Anand, Venktesh V. Deepali Prabhu","submitted_at":"2024-06-24T22:09:50Z","abstract_excerpt":"Open-domain complex Question Answering (QA) is a difficult task with challenges in evidence retrieval and reasoning. The complexity of such questions could stem from questions being compositional, hybrid evidence, or ambiguity in questions. While retrieval performance for classical QA tasks is well explored, their capabilities for heterogeneous complex retrieval tasks, especially in an open-domain setting, and the impact on downstream QA performance, are relatively unexplored. To address this, in this work, we propose a benchmark composing diverse complex QA tasks and provide a toolkit to eval"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.17158","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-06-24T22:09:50Z","cross_cats_sorted":["cs.IR"],"title_canon_sha256":"14c602c041600d5fbc0d9d18b4b00f505af390e081959589bd4536da4f3ddcea","abstract_canon_sha256":"49c4b300a2b7d425928f744edeacfd258dca34f368bbcaf81e94a3b23f677267"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:36:30.459455Z","signature_b64":"rT+0SdYgRIGWMRrUrqRooUSf8d4VnaUUZECzPjh1Wt58x1tT7xxjsYJrjqBNrIGRkp4lfASgTMw2klYSuHYtDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"39af3900997d4a49f2d6fd017cd20da302b453542dff5d2831d3fbea3ecd4d12","last_reissued_at":"2026-07-05T08:36:30.458933Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:36:30.458933Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DEXTER: A Benchmark for open-domain Complex Question Answering using LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.IR"],"primary_cat":"cs.CL","authors_text":"Avishek Anand, Venktesh V. Deepali Prabhu","submitted_at":"2024-06-24T22:09:50Z","abstract_excerpt":"Open-domain complex Question Answering (QA) is a difficult task with challenges in evidence retrieval and reasoning. The complexity of such questions could stem from questions being compositional, hybrid evidence, or ambiguity in questions. While retrieval performance for classical QA tasks is well explored, their capabilities for heterogeneous complex retrieval tasks, especially in an open-domain setting, and the impact on downstream QA performance, are relatively unexplored. To address this, in this work, we propose a benchmark composing diverse complex QA tasks and provide a toolkit to eval"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.17158","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.17158/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.17158","created_at":"2026-07-05T08:36:30.458990+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.17158v1","created_at":"2026-07-05T08:36:30.458990+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.17158","created_at":"2026-07-05T08:36:30.458990+00:00"},{"alias_kind":"pith_short_12","alias_value":"HGXTSAEZPVFE","created_at":"2026-07-05T08:36:30.458990+00:00"},{"alias_kind":"pith_short_16","alias_value":"HGXTSAEZPVFET4WW","created_at":"2026-07-05T08:36:30.458990+00:00"},{"alias_kind":"pith_short_8","alias_value":"HGXTSAEZ","created_at":"2026-07-05T08:36:30.458990+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.12694","citing_title":"LLM-based Query Expansion Fails for Unfamiliar and Ambiguous Queries","ref_index":27,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HGXTSAEZPVFET4WW7UAXZUQNUM","json":"https://pith.science/pith/HGXTSAEZPVFET4WW7UAXZUQNUM.json","graph_json":"https://pith.science/api/pith-number/HGXTSAEZPVFET4WW7UAXZUQNUM/graph.json","events_json":"https://pith.science/api/pith-number/HGXTSAEZPVFET4WW7UAXZUQNUM/events.json","paper":"https://pith.science/paper/HGXTSAEZ"},"agent_actions":{"view_html":"https://pith.science/pith/HGXTSAEZPVFET4WW7UAXZUQNUM","download_json":"https://pith.science/pith/HGXTSAEZPVFET4WW7UAXZUQNUM.json","view_paper":"https://pith.science/paper/HGXTSAEZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.17158&json=true","fetch_graph":"https://pith.science/api/pith-number/HGXTSAEZPVFET4WW7UAXZUQNUM/graph.json","fetch_events":"https://pith.science/api/pith-number/HGXTSAEZPVFET4WW7UAXZUQNUM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HGXTSAEZPVFET4WW7UAXZUQNUM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HGXTSAEZPVFET4WW7UAXZUQNUM/action/storage_attestation","attest_author":"https://pith.science/pith/HGXTSAEZPVFET4WW7UAXZUQNUM/action/author_attestation","sign_citation":"https://pith.science/pith/HGXTSAEZPVFET4WW7UAXZUQNUM/action/citation_signature","submit_replication":"https://pith.science/pith/HGXTSAEZPVFET4WW7UAXZUQNUM/action/replication_record"}},"created_at":"2026-07-05T08:36:30.458990+00:00","updated_at":"2026-07-05T08:36:30.458990+00:00"}