{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:K4V4TQUPK7V5HTZIPEP5LDF4FI","short_pith_number":"pith:K4V4TQUP","schema_version":"1.0","canonical_sha256":"572bc9c28f57ebd3cf28791fd58cbc2a2db2eb2cfa68b4d6ee5ff316325e05a5","source":{"kind":"arxiv","id":"2501.13080","version":1},"attestation_state":"computed","paper":{"title":"Refining Input Guardrails: Enhancing LLM-as-a-Judge Efficiency Through Chain-of-Thought Fine-Tuning and Alignment","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Andy Luo, Huy Nghiem, Melissa Kazemi Rad, Mohammad Sorower, Sahil Wadhwa, Stephen Rawls","submitted_at":"2025-01-22T18:40:57Z","abstract_excerpt":"Large Language Models (LLMs) have demonstrated powerful capabilities that render them valuable in different applications, including conversational AI products. It is paramount to ensure the security and reliability of these products by mitigating their vulnerabilities towards malicious user interactions, which can lead to the exposure of great risks and reputational repercussions. In this work, we present a comprehensive study on the efficacy of fine-tuning and aligning Chain-of-Thought (CoT) responses of different LLMs that serve as input moderation guardrails. We systematically explore vario"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.13080","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-01-22T18:40:57Z","cross_cats_sorted":["cs.CR","cs.LG"],"title_canon_sha256":"d63cbe8f72a55f5f76734b14b9133cdeafc7e2566e131c3359389fff681ab014","abstract_canon_sha256":"f29cd52084f8b2d741116f0655bb75c424ce3d6949e42e89608216dc5ebaa41c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:04:07.222582Z","signature_b64":"eNyIR78rT7pljKGqKL6VlVld/7irKQ7Uz/5rke+ibY5cl9Fa0gD2VvREpCfi1oupSsdWaa6dKzWxrd9NHULUAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"572bc9c28f57ebd3cf28791fd58cbc2a2db2eb2cfa68b4d6ee5ff316325e05a5","last_reissued_at":"2026-07-05T10:04:07.222094Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:04:07.222094Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Refining Input Guardrails: Enhancing LLM-as-a-Judge Efficiency Through Chain-of-Thought Fine-Tuning and Alignment","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Andy Luo, Huy Nghiem, Melissa Kazemi Rad, Mohammad Sorower, Sahil Wadhwa, Stephen Rawls","submitted_at":"2025-01-22T18:40:57Z","abstract_excerpt":"Large Language Models (LLMs) have demonstrated powerful capabilities that render them valuable in different applications, including conversational AI products. It is paramount to ensure the security and reliability of these products by mitigating their vulnerabilities towards malicious user interactions, which can lead to the exposure of great risks and reputational repercussions. In this work, we present a comprehensive study on the efficacy of fine-tuning and aligning Chain-of-Thought (CoT) responses of different LLMs that serve as input moderation guardrails. We systematically explore vario"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.13080","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.13080/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.13080","created_at":"2026-07-05T10:04:07.222152+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.13080v1","created_at":"2026-07-05T10:04:07.222152+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.13080","created_at":"2026-07-05T10:04:07.222152+00:00"},{"alias_kind":"pith_short_12","alias_value":"K4V4TQUPK7V5","created_at":"2026-07-05T10:04:07.222152+00:00"},{"alias_kind":"pith_short_16","alias_value":"K4V4TQUPK7V5HTZI","created_at":"2026-07-05T10:04:07.222152+00:00"},{"alias_kind":"pith_short_8","alias_value":"K4V4TQUP","created_at":"2026-07-05T10:04:07.222152+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2510.13727","citing_title":"From Refusal to Recovery: A Control-Theoretic Approach to Generative AI Guardrails","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2509.15174","citing_title":"SMARTER: A Data-efficient Framework to Improve Toxicity Detection with Explanation via Self-augmenting Large Language Models","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/K4V4TQUPK7V5HTZIPEP5LDF4FI","json":"https://pith.science/pith/K4V4TQUPK7V5HTZIPEP5LDF4FI.json","graph_json":"https://pith.science/api/pith-number/K4V4TQUPK7V5HTZIPEP5LDF4FI/graph.json","events_json":"https://pith.science/api/pith-number/K4V4TQUPK7V5HTZIPEP5LDF4FI/events.json","paper":"https://pith.science/paper/K4V4TQUP"},"agent_actions":{"view_html":"https://pith.science/pith/K4V4TQUPK7V5HTZIPEP5LDF4FI","download_json":"https://pith.science/pith/K4V4TQUPK7V5HTZIPEP5LDF4FI.json","view_paper":"https://pith.science/paper/K4V4TQUP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.13080&json=true","fetch_graph":"https://pith.science/api/pith-number/K4V4TQUPK7V5HTZIPEP5LDF4FI/graph.json","fetch_events":"https://pith.science/api/pith-number/K4V4TQUPK7V5HTZIPEP5LDF4FI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/K4V4TQUPK7V5HTZIPEP5LDF4FI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/K4V4TQUPK7V5HTZIPEP5LDF4FI/action/storage_attestation","attest_author":"https://pith.science/pith/K4V4TQUPK7V5HTZIPEP5LDF4FI/action/author_attestation","sign_citation":"https://pith.science/pith/K4V4TQUPK7V5HTZIPEP5LDF4FI/action/citation_signature","submit_replication":"https://pith.science/pith/K4V4TQUPK7V5HTZIPEP5LDF4FI/action/replication_record"}},"created_at":"2026-07-05T10:04:07.222152+00:00","updated_at":"2026-07-05T10:04:07.222152+00:00"}