{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:GY47AMURSOTJOK75NEKOBS2RF7","short_pith_number":"pith:GY47AMUR","schema_version":"1.0","canonical_sha256":"3639f0329193a6972bfd6914e0cb512fe85c4605f35230e109e798aa88f5ccb9","source":{"kind":"arxiv","id":"2310.00752","version":4},"attestation_state":"computed","paper":{"title":"TIGERScore: Towards Building Explainable Metric for All Text Generation Tasks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Bill Yuchen Lin, Dongfu Jiang, Ge Zhang, Wenhao Huang, Wenhu Chen, Yishan Li","submitted_at":"2023-10-01T18:01:51Z","abstract_excerpt":"We present TIGERScore, a \\textbf{T}rained metric that follows \\textbf{I}nstruction \\textbf{G}uidance to perform \\textbf{E}xplainable, and \\textbf{R}eference-free evaluation over a wide spectrum of text generation tasks. Different from other automatic evaluation methods that only provide arcane scores, TIGERScore is guided by natural language instruction to provide error analysis to pinpoint the mistakes in the generated text. Our metric is based on LLaMA-2, trained on our meticulously curated instruction-tuning dataset MetricInstruct which covers 6 text generation tasks and 23 text generation "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.00752","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-10-01T18:01:51Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"fb54331c5262b2bdbc0e3d25c7d46bec83353546ec965cc6acaa68fc02415a05","abstract_canon_sha256":"0ca046ff4e15bd9d03c1908e4d70913f46ecea02d468e71670945bc05eb71542"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:17:28.828031Z","signature_b64":"pqsg1zUHR1bjPJDvSpEC4JaFIIAoWx/J96r9AdKOXkYIfh700N/ShB1ruSxlPT6iUxYUQqv1TAYa+eoQyEseCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3639f0329193a6972bfd6914e0cb512fe85c4605f35230e109e798aa88f5ccb9","last_reissued_at":"2026-07-05T08:17:28.827530Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:17:28.827530Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TIGERScore: Towards Building Explainable Metric for All Text Generation Tasks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Bill Yuchen Lin, Dongfu Jiang, Ge Zhang, Wenhao Huang, Wenhu Chen, Yishan Li","submitted_at":"2023-10-01T18:01:51Z","abstract_excerpt":"We present TIGERScore, a \\textbf{T}rained metric that follows \\textbf{I}nstruction \\textbf{G}uidance to perform \\textbf{E}xplainable, and \\textbf{R}eference-free evaluation over a wide spectrum of text generation tasks. Different from other automatic evaluation methods that only provide arcane scores, TIGERScore is guided by natural language instruction to provide error analysis to pinpoint the mistakes in the generated text. Our metric is based on LLaMA-2, trained on our meticulously curated instruction-tuning dataset MetricInstruct which covers 6 text generation tasks and 23 text generation "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.00752","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.00752/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.00752","created_at":"2026-07-05T08:17:28.827600+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.00752v4","created_at":"2026-07-05T08:17:28.827600+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.00752","created_at":"2026-07-05T08:17:28.827600+00:00"},{"alias_kind":"pith_short_12","alias_value":"GY47AMURSOTJ","created_at":"2026-07-05T08:17:28.827600+00:00"},{"alias_kind":"pith_short_16","alias_value":"GY47AMURSOTJOK75","created_at":"2026-07-05T08:17:28.827600+00:00"},{"alias_kind":"pith_short_8","alias_value":"GY47AMUR","created_at":"2026-07-05T08:17:28.827600+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08268","citing_title":"Different Teachers, Different Capabilities: Sub-1B On-Device Distillation for Structured Text Enrichment","ref_index":42,"is_internal_anchor":true},{"citing_arxiv_id":"2606.00093","citing_title":"Agreement Metrics for LLM-as-Judge Evaluation: What to Report and Why","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09492","citing_title":"APCD: Adaptive Path-Contrastive Decoding for Reliable Large Language Model Generation","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23178","citing_title":"Judging the Judges: A Systematic Evaluation of Bias Mitigation Strategies in LLM-as-a-Judge Pipelines","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GY47AMURSOTJOK75NEKOBS2RF7","json":"https://pith.science/pith/GY47AMURSOTJOK75NEKOBS2RF7.json","graph_json":"https://pith.science/api/pith-number/GY47AMURSOTJOK75NEKOBS2RF7/graph.json","events_json":"https://pith.science/api/pith-number/GY47AMURSOTJOK75NEKOBS2RF7/events.json","paper":"https://pith.science/paper/GY47AMUR"},"agent_actions":{"view_html":"https://pith.science/pith/GY47AMURSOTJOK75NEKOBS2RF7","download_json":"https://pith.science/pith/GY47AMURSOTJOK75NEKOBS2RF7.json","view_paper":"https://pith.science/paper/GY47AMUR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.00752&json=true","fetch_graph":"https://pith.science/api/pith-number/GY47AMURSOTJOK75NEKOBS2RF7/graph.json","fetch_events":"https://pith.science/api/pith-number/GY47AMURSOTJOK75NEKOBS2RF7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GY47AMURSOTJOK75NEKOBS2RF7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GY47AMURSOTJOK75NEKOBS2RF7/action/storage_attestation","attest_author":"https://pith.science/pith/GY47AMURSOTJOK75NEKOBS2RF7/action/author_attestation","sign_citation":"https://pith.science/pith/GY47AMURSOTJOK75NEKOBS2RF7/action/citation_signature","submit_replication":"https://pith.science/pith/GY47AMURSOTJOK75NEKOBS2RF7/action/replication_record"}},"created_at":"2026-07-05T08:17:28.827600+00:00","updated_at":"2026-07-05T08:17:28.827600+00:00"}