{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:AMF2ZA2ANYVR4IPVMNEGMAKZPY","short_pith_number":"pith:AMF2ZA2A","schema_version":"1.0","canonical_sha256":"030bac83406e2b1e21f563486601597e02c5069a03f9e1f7b92fc1ca5ed590cb","source":{"kind":"arxiv","id":"2310.03269","version":1},"attestation_state":"computed","paper":{"title":"InstructProtein: Aligning Human and Protein Language via Knowledge Instruction","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"q-bio.BM","authors_text":"Huajun Chen, Keyan Ding, Ming Qin, Qiang Zhang, Xiang Zhuang, Xiaotong Li, Zeyuan Wang","submitted_at":"2023-10-05T02:45:39Z","abstract_excerpt":"Large Language Models (LLMs) have revolutionized the field of natural language processing, but they fall short in comprehending biological sequences such as proteins. To address this challenge, we propose InstructProtein, an innovative LLM that possesses bidirectional generation capabilities in both human and protein languages: (i) taking a protein sequence as input to predict its textual function description and (ii) using natural language to prompt protein sequence generation. To achieve this, we first pre-train an LLM on both protein and natural language corpora, enabling it to comprehend i"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.03269","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"q-bio.BM","submitted_at":"2023-10-05T02:45:39Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"fb9cfb3b69d0be36618ab91a7b3d160b8dcf05f7e9064ab70174263a30ec0079","abstract_canon_sha256":"052f8d5e698bb422e4700746685b35b2b822c039c12fab4977423182f7715249"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:57:30.856990Z","signature_b64":"4Xj9Yk4sxdGAYeunyLUm3yndb5iO6wkXfAb+aGfWVdqAbS2fy5hsKo9lxEJblsi715OkR+mSSPOuD4FHa199AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"030bac83406e2b1e21f563486601597e02c5069a03f9e1f7b92fc1ca5ed590cb","last_reissued_at":"2026-07-05T06:57:30.856544Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:57:30.856544Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"InstructProtein: Aligning Human and Protein Language via Knowledge Instruction","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"q-bio.BM","authors_text":"Huajun Chen, Keyan Ding, Ming Qin, Qiang Zhang, Xiang Zhuang, Xiaotong Li, Zeyuan Wang","submitted_at":"2023-10-05T02:45:39Z","abstract_excerpt":"Large Language Models (LLMs) have revolutionized the field of natural language processing, but they fall short in comprehending biological sequences such as proteins. To address this challenge, we propose InstructProtein, an innovative LLM that possesses bidirectional generation capabilities in both human and protein languages: (i) taking a protein sequence as input to predict its textual function description and (ii) using natural language to prompt protein sequence generation. To achieve this, we first pre-train an LLM on both protein and natural language corpora, enabling it to comprehend i"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.03269","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.03269/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.03269","created_at":"2026-07-05T06:57:30.856606+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.03269v1","created_at":"2026-07-05T06:57:30.856606+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.03269","created_at":"2026-07-05T06:57:30.856606+00:00"},{"alias_kind":"pith_short_12","alias_value":"AMF2ZA2ANYVR","created_at":"2026-07-05T06:57:30.856606+00:00"},{"alias_kind":"pith_short_16","alias_value":"AMF2ZA2ANYVR4IPV","created_at":"2026-07-05T06:57:30.856606+00:00"},{"alias_kind":"pith_short_8","alias_value":"AMF2ZA2A","created_at":"2026-07-05T06:57:30.856606+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2506.03800","citing_title":"STELLA: A Multimodal LLM for Protein Functional Annotation via Unified Sequence-Structure Encoding","ref_index":18,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AMF2ZA2ANYVR4IPVMNEGMAKZPY","json":"https://pith.science/pith/AMF2ZA2ANYVR4IPVMNEGMAKZPY.json","graph_json":"https://pith.science/api/pith-number/AMF2ZA2ANYVR4IPVMNEGMAKZPY/graph.json","events_json":"https://pith.science/api/pith-number/AMF2ZA2ANYVR4IPVMNEGMAKZPY/events.json","paper":"https://pith.science/paper/AMF2ZA2A"},"agent_actions":{"view_html":"https://pith.science/pith/AMF2ZA2ANYVR4IPVMNEGMAKZPY","download_json":"https://pith.science/pith/AMF2ZA2ANYVR4IPVMNEGMAKZPY.json","view_paper":"https://pith.science/paper/AMF2ZA2A","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.03269&json=true","fetch_graph":"https://pith.science/api/pith-number/AMF2ZA2ANYVR4IPVMNEGMAKZPY/graph.json","fetch_events":"https://pith.science/api/pith-number/AMF2ZA2ANYVR4IPVMNEGMAKZPY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AMF2ZA2ANYVR4IPVMNEGMAKZPY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AMF2ZA2ANYVR4IPVMNEGMAKZPY/action/storage_attestation","attest_author":"https://pith.science/pith/AMF2ZA2ANYVR4IPVMNEGMAKZPY/action/author_attestation","sign_citation":"https://pith.science/pith/AMF2ZA2ANYVR4IPVMNEGMAKZPY/action/citation_signature","submit_replication":"https://pith.science/pith/AMF2ZA2ANYVR4IPVMNEGMAKZPY/action/replication_record"}},"created_at":"2026-07-05T06:57:30.856606+00:00","updated_at":"2026-07-05T06:57:30.856606+00:00"}