{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:7DRKA2LHJCADJWAMQ63KAGC4GF","short_pith_number":"pith:7DRKA2LH","schema_version":"1.0","canonical_sha256":"f8e2a06967488034d80c87b6a0185c317903f6916b62346b6c6a5340af15dfc7","source":{"kind":"arxiv","id":"2505.16366","version":1},"attestation_state":"computed","paper":{"title":"ReCopilot: Reverse Engineering Copilot in Binary Analysis","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CR","authors_text":"Bin Yin, Daguang Liu, Guoqiang Chen, Huiqi Sun, Lingyun Ying, Lu Liu, Qiang Wang, Zhiqi Wang","submitted_at":"2025-05-22T08:21:39Z","abstract_excerpt":"Binary analysis plays a pivotal role in security domains such as malware detection and vulnerability discovery, yet it remains labor-intensive and heavily reliant on expert knowledge. General-purpose large language models (LLMs) perform well in programming analysis on source code, while binaryspecific LLMs are underexplored. In this work, we present ReCopilot, an expert LLM designed for binary analysis tasks. ReCopilot integrates binary code knowledge through a meticulously constructed dataset, encompassing continue pretraining (CPT), supervised fine-tuning (SFT), and direct preference optimiz"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.16366","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CR","submitted_at":"2025-05-22T08:21:39Z","cross_cats_sorted":[],"title_canon_sha256":"21d6ac8be2910b576468a6636c755da45e044aa20d7a096e350665e24de2b25a","abstract_canon_sha256":"61f01bbe8a33bf7ebed6b046ab8afce203533f5e1e774f1e5811569d3c3eb52d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:07:34.241991Z","signature_b64":"g7azLgJCwnvMJoV7uXfjiUpIC2D9b9hsvcYWtKYRzaDi0gF0voQv6F4mhKeFaMSLsAmafzhlonAgU17nlXpxDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f8e2a06967488034d80c87b6a0185c317903f6916b62346b6c6a5340af15dfc7","last_reissued_at":"2026-07-05T11:07:34.241394Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:07:34.241394Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ReCopilot: Reverse Engineering Copilot in Binary Analysis","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CR","authors_text":"Bin Yin, Daguang Liu, Guoqiang Chen, Huiqi Sun, Lingyun Ying, Lu Liu, Qiang Wang, Zhiqi Wang","submitted_at":"2025-05-22T08:21:39Z","abstract_excerpt":"Binary analysis plays a pivotal role in security domains such as malware detection and vulnerability discovery, yet it remains labor-intensive and heavily reliant on expert knowledge. General-purpose large language models (LLMs) perform well in programming analysis on source code, while binaryspecific LLMs are underexplored. In this work, we present ReCopilot, an expert LLM designed for binary analysis tasks. ReCopilot integrates binary code knowledge through a meticulously constructed dataset, encompassing continue pretraining (CPT), supervised fine-tuning (SFT), and direct preference optimiz"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.16366","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.16366/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.16366","created_at":"2026-07-05T11:07:34.241470+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.16366v1","created_at":"2026-07-05T11:07:34.241470+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.16366","created_at":"2026-07-05T11:07:34.241470+00:00"},{"alias_kind":"pith_short_12","alias_value":"7DRKA2LHJCAD","created_at":"2026-07-05T11:07:34.241470+00:00"},{"alias_kind":"pith_short_16","alias_value":"7DRKA2LHJCADJWAM","created_at":"2026-07-05T11:07:34.241470+00:00"},{"alias_kind":"pith_short_8","alias_value":"7DRKA2LH","created_at":"2026-07-05T11:07:34.241470+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.08083","citing_title":"Can LLMs Deobfuscate Binary Code? A Systematic Analysis of Large Language Models into Pseudocode Deobfuscation","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14317","citing_title":"Challenges and Future Directions in Agentic Reverse Engineering Systems","ref_index":32,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7DRKA2LHJCADJWAMQ63KAGC4GF","json":"https://pith.science/pith/7DRKA2LHJCADJWAMQ63KAGC4GF.json","graph_json":"https://pith.science/api/pith-number/7DRKA2LHJCADJWAMQ63KAGC4GF/graph.json","events_json":"https://pith.science/api/pith-number/7DRKA2LHJCADJWAMQ63KAGC4GF/events.json","paper":"https://pith.science/paper/7DRKA2LH"},"agent_actions":{"view_html":"https://pith.science/pith/7DRKA2LHJCADJWAMQ63KAGC4GF","download_json":"https://pith.science/pith/7DRKA2LHJCADJWAMQ63KAGC4GF.json","view_paper":"https://pith.science/paper/7DRKA2LH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.16366&json=true","fetch_graph":"https://pith.science/api/pith-number/7DRKA2LHJCADJWAMQ63KAGC4GF/graph.json","fetch_events":"https://pith.science/api/pith-number/7DRKA2LHJCADJWAMQ63KAGC4GF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7DRKA2LHJCADJWAMQ63KAGC4GF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7DRKA2LHJCADJWAMQ63KAGC4GF/action/storage_attestation","attest_author":"https://pith.science/pith/7DRKA2LHJCADJWAMQ63KAGC4GF/action/author_attestation","sign_citation":"https://pith.science/pith/7DRKA2LHJCADJWAMQ63KAGC4GF/action/citation_signature","submit_replication":"https://pith.science/pith/7DRKA2LHJCADJWAMQ63KAGC4GF/action/replication_record"}},"created_at":"2026-07-05T11:07:34.241470+00:00","updated_at":"2026-07-05T11:07:34.241470+00:00"}