From 903ee9f90ee4df8ee2057019dc51f69a5b13e34e Mon Sep 17 00:00:00 2001 From: norsage Date: Sat, 26 Sep 2026 01:44:42 +0300 Subject: [PATCH 1/5] fix(ui): satisfy clippy collapsible_match in object actions Move the materialize precondition into a match guard. When the guard fails, the action falls through to the existing no-op arm, as before. Co-Authored-By: Claude Opus 5.5 --- patinae/src/bridges/objects.rs | 26 ++++++++++++-------------- 1 file changed, 12 insertions(+), 14 deletions(-) diff --git a/patinae/src/bridges/objects.rs b/patinae/src/bridges/objects.rs index db07d3b..ba9c220 100644 --- a/patinae/src/bridges/objects.rs +++ b/patinae/src/bridges/objects.rs @@ -2537,22 +2537,20 @@ pub fn setup_callbacks(app: Rc>, window: &AppWindow) { let short = ObjectsBridge::truncate_name(&target, 28); match action.as_str() { - "materialize" => { + "materialize" if a.objects.selection_level == SelectionLevel::Objects - && a.objects.selected_objects.len() == 1 + && a.objects.selected_objects.len() == 1 => + { + let name = a.objects.selected_objects[0].clone(); + if a.kernel + .scene + .get(&name) + .is_some_and(|object| object.can_materialize) { - let name = a.objects.selected_objects[0].clone(); - if a.kernel - .scene - .get(&name) - .is_some_and(|object| object.can_materialize) - { - a.kernel.bus.execute_command(format!( - "materialize {}", - quote_command_arg(&name) - )); - os.set_popover_kind("".into()); - } + a.kernel + .bus + .execute_command(format!("materialize {}", quote_command_arg(&name))); + os.set_popover_kind("".into()); } } "rename" if capabilities.rename => open_name_popup(&os, "rename", &target, &short), From b76fae50f3a965c01c397d35a37e78c31a284c69 Mon Sep 17 00:00:00 2001 From: norsage Date: Sat, 26 Sep 2026 12:57:56 +0300 Subject: [PATCH 2/5] perf(io): reuse the previous atom's residue when building molecules Atoms of one residue are contiguous in coordinate files, so an atom whose residue fields match the previous atom's reuses its Arc without building and hashing a cache key. The cache still shares non-contiguous atoms of one residue. On 487 PDB mmCIF files (5.2M atoms), instructions retired for reading them drop from 86.5G to 75.9G. Co-Authored-By: Claude Opus 5.5 --- crates/patinae-io/src/logical_models.rs | 67 +++++++++++++++++++++---- 1 file changed, 56 insertions(+), 11 deletions(-) diff --git a/crates/patinae-io/src/logical_models.rs b/crates/patinae-io/src/logical_models.rs index 5f4c28f..2af849b 100644 --- a/crates/patinae-io/src/logical_models.rs +++ b/crates/patinae-io/src/logical_models.rs @@ -24,6 +24,27 @@ pub(crate) struct ParsedAtom { pub(crate) segi: String, } +impl ParsedAtom { + fn residue(&self) -> AtomResidue { + AtomResidue::from_parts( + self.chain.clone(), + self.resn.clone(), + self.resv, + self.icode, + self.segi.clone(), + ) + } + + /// Whether [`ParsedAtom::residue`] gives equal residues for both atoms. + fn same_residue(&self, other: &ParsedAtom) -> bool { + self.chain == other.chain + && self.resn == other.resn + && self.resv == other.resv + && self.icode == other.icode + && self.segi == other.segi + } +} + #[derive(Debug, Clone)] pub(crate) struct ParsedModel { pub(crate) model_number: i32, @@ -218,19 +239,21 @@ fn build_group_molecule( } let mut residue_cache: HashMap> = HashMap::new(); + let mut previous: Option<(&ParsedAtom, Arc)> = None; for parsed in &first_model.atoms { let mut atom = Atom::new(parsed.name.as_str(), parsed.element); - let residue_data = AtomResidue::from_parts( - parsed.chain.clone(), - parsed.resn.clone(), - parsed.resv, - parsed.icode, - parsed.segi.clone(), - ); - atom.residue = residue_cache - .entry(residue_data.clone()) - .or_insert_with(|| Arc::new(residue_data)) - .clone(); + atom.residue = match &previous { + // Atoms of one residue are contiguous: skip building and hashing its key. + Some((last, residue)) if last.same_residue(parsed) => residue.clone(), + _ => { + let residue_data = parsed.residue(); + residue_cache + .entry(residue_data.clone()) + .or_insert_with(|| Arc::new(residue_data)) + .clone() + } + }; + previous = Some((parsed, atom.residue.clone())); atom.alt = parsed.alt; atom.b_factor = parsed.b_factor; atom.occupancy = parsed.occupancy; @@ -306,4 +329,26 @@ mod tests { assert_eq!(molecules[0].state_count(), 1); assert_eq!(molecules[0].atom_count(), 1); } + + #[test] + fn residues_are_shared_by_value_not_by_adjacency() { + let mut model = model(); + let template = model.atoms[0].clone(); + for resv in [1, 2, 1] { + model.atoms.push(ParsedAtom { + resv, + ..template.clone() + }); + model.coords.push(Vec3::new(0.0, 0.0, 0.0)); + } + + let molecule = build_molecules("mol", "", vec![model]).unwrap().remove(0); + let residues: Vec<_> = molecule.atoms().map(|atom| &atom.residue).collect(); + + assert!(Arc::ptr_eq(residues[0], residues[1])); + assert!(!Arc::ptr_eq(residues[1], residues[2])); + // Non-contiguous atoms of one residue still share it. + assert!(Arc::ptr_eq(residues[0], residues[3])); + assert!(!Arc::ptr_eq(residues[2], residues[3])); + } } From d8a0865cee7889ce2a6745ad8c405e348741e4ef Mon Sep 17 00:00:00 2001 From: norsage Date: Sat, 26 Sep 2026 12:59:08 +0300 Subject: [PATCH 3/5] feat(io): keep mmCIF label residue ids per residue Store label_asym_id, label_entity_id and label_seq_id on AtomResidue when reading mmCIF and bCIF. AF3-style models number residues by the label scheme, and label_seq_id is what links atoms to _entity_poly_seq. Previously label_asym_id survived only in assembly membership and label_entity_id was not read. bCIF encoders store entity ids either as strings or as integers; both are read. The fields are None for PDB and other formats, and default to None when older sessions are deserialized. Topology grouping of models still uses auth identity only, so grouped models keep the first model's labels. Label ids describe the file as loaded: edits do not renumber them, and build_mutant now clones the template residue so they survive mutation. Label strings are shared with the previous atom when equal, so parsing does not allocate them per atom. Co-Authored-By: Claude Opus 5.5 --- crates/patinae-io/src/bcif/parser.rs | 59 +++++++++++++-- crates/patinae-io/src/bcif/test_support.rs | 2 +- crates/patinae-io/src/cif/parser.rs | 86 +++++++++++++++++++++- crates/patinae-io/src/logical_models.rs | 78 ++++++++++++++++++-- crates/patinae-io/src/pdb/parser.rs | 16 +++- crates/patinae-mm/src/mutate.rs | 56 ++++++++++++-- crates/patinae-mol/src/atom.rs | 49 +++++++++--- 7 files changed, 316 insertions(+), 30 deletions(-) diff --git a/crates/patinae-io/src/bcif/parser.rs b/crates/patinae-io/src/bcif/parser.rs index 70142d4..c020360 100644 --- a/crates/patinae-io/src/bcif/parser.rs +++ b/crates/patinae-io/src/bcif/parser.rs @@ -2,6 +2,7 @@ //! //! Parses BinaryCIF format files into ObjectMolecule. +use std::borrow::Cow; use std::collections::{BTreeMap, HashMap}; use std::io::Read; @@ -11,7 +12,7 @@ use patinae_mol::{Element, ObjectMolecule, SecondaryStructure}; use crate::assembly::{AssemblyRow, AssemblyRows, CATEGORIES}; use crate::cif::common::{apply_secondary_structure, SecondaryStructureRange, SsCategory}; use crate::error::{IoError, IoResult}; -use crate::logical_models::{build_molecules, ParsedAtom, ParsedModel}; +use crate::logical_models::{build_molecules, ParsedAtom, ParsedLabels, ParsedModel}; use crate::traits::MoleculeReader; use super::decode::{decode_column, decode_mask, ColumnMask, DecodedColumn}; @@ -103,6 +104,14 @@ impl CategoryColumns { col.str_at(i).filter(|s| !s.is_empty()) } + /// Text of a column that encoders store either as strings or as integers. + fn code_at(&self, name: &str, i: usize) -> Option> { + self.str_at(name, i).map(Cow::Borrowed).or_else(|| { + self.int_at(name, i) + .map(|value| Cow::Owned(value.to_string())) + }) + } + fn int_at(&self, name: &str, i: usize) -> Option { let (col, mask) = self.columns.get(name)?; if let Some(m) = mask { @@ -273,6 +282,14 @@ fn parse_atom_site( // become "A2", which is valid in the selection language. let chain_id = chain.replace('-', ""); + // mmCIF label residue ids, kept alongside the auth ids above + let label_asym_id = cols.str_at("label_asym_id", i); + let label_entity_id = cols.code_at("label_entity_id", i); + let label_seq_id = cols + .int_at("label_seq_id", i) + .and_then(|seq_id| u32::try_from(seq_id).ok()) + .filter(|&seq_id| seq_id > 0); + // Alt loc let alt = cols .str_at("label_alt_id", i) @@ -310,9 +327,13 @@ fn parse_atom_site( let model = models .entry(model_num) .or_insert_with(|| ParsedModel::new(model_num)); - model - .source_chains - .push(cols.str_at("label_asym_id", i).unwrap_or("")); + model.source_chains.push(label_asym_id.unwrap_or("")); + let labels = ParsedLabels::new( + label_asym_id, + label_entity_id.as_deref(), + label_seq_id, + model.atoms.last(), + ); model.atoms.push(ParsedAtom { name: atom_name.to_string(), element, @@ -327,6 +348,7 @@ fn parse_atom_site( occupancy, b_factor, segi: String::new(), + labels, }); model.coords.push(Vec3::new(x, y, z)); } @@ -493,7 +515,9 @@ fn parse_entry(category: &BcifCategory, title: &mut String) -> IoResult<()> { #[cfg(test)] mod tests { use super::*; - use crate::bcif::test_support::{atom_row, atom_site_block, bcif_file, encode_bcif_file}; + use crate::bcif::test_support::{ + atom_row, atom_site_block, bcif_file, encode_bcif_file, int_column, string_column, + }; use std::collections::HashMap; /// Verify that CategoryColumns::str_at filters out empty strings, @@ -613,4 +637,29 @@ mod tests { assert_eq!(molecule.state_count(), 1); assert_eq!(molecule.atoms_slice()[0].residue.chain, "A"); } + + #[test] + fn label_fields_are_kept_alongside_auth_fields() { + let rows = [ + atom_row(1, "N", "ALA", "B", 0.0, 1), + atom_row(2, "CA", "ALA", "B", 1.5, 1), + ]; + let mut block = atom_site_block("AF3", &rows); + block.categories[0].columns.extend([ + string_column("auth_asym_id", ["H", "H"].into_iter()), + int_column("auth_seq_id", [101, 101].into_iter()), + int_column("label_entity_id", [2, 2].into_iter()), + ]); + + let molecule = + crate::bcif::read_bcif_bytes_with_bond_tolerance(&encode_bcif_file(&[block]), 0.1) + .unwrap(); + let atoms = molecule.atoms_slice(); + let residue = &atoms[0].residue; + + assert_eq!((residue.chain.as_str(), residue.resv), ("H", 101)); + assert_eq!(residue.label_asym_id.as_deref(), Some("B")); + assert_eq!(residue.label_entity_id.as_deref(), Some("2")); + assert_eq!(residue.label_seq_id, Some(1)); + } } diff --git a/crates/patinae-io/src/bcif/test_support.rs b/crates/patinae-io/src/bcif/test_support.rs index ee66bf5..02e3110 100644 --- a/crates/patinae-io/src/bcif/test_support.rs +++ b/crates/patinae-io/src/bcif/test_support.rs @@ -95,7 +95,7 @@ pub(crate) fn encode_bcif_file(blocks: &[BcifDataBlock]) -> Vec { rmp_serde::to_vec_named(&file).expect("test bCIF fixture should serialize") } -fn int_column(name: &str, values: impl Iterator) -> BcifColumn { +pub(crate) fn int_column(name: &str, values: impl Iterator) -> BcifColumn { BcifColumn { name: name.to_string(), data: BcifData { diff --git a/crates/patinae-io/src/cif/parser.rs b/crates/patinae-io/src/cif/parser.rs index 1561072..7c61ed6 100644 --- a/crates/patinae-io/src/cif/parser.rs +++ b/crates/patinae-io/src/cif/parser.rs @@ -10,7 +10,7 @@ use patinae_mol::{Element, ObjectMolecule}; use crate::assembly::{AssemblyRow, AssemblyRows, CATEGORIES}; use crate::error::{IoError, IoResult}; -use crate::logical_models::{build_molecules, ParsedAtom, ParsedModel}; +use crate::logical_models::{build_molecules, ParsedAtom, ParsedLabels, ParsedModel}; use crate::traits::MoleculeReader; use super::common::{ @@ -30,6 +30,7 @@ struct AtomSiteColumns { auth_asym_id: Option, label_seq_id: Option, auth_seq_id: Option, + label_entity_id: Option, pdbx_pdb_ins_code: Option, label_alt_id: Option, b_iso_or_equiv: Option, @@ -126,6 +127,7 @@ impl AtomSiteColumns { "_atom_site.auth_asym_id" => cols.auth_asym_id = Some(i), "_atom_site.label_seq_id" => cols.label_seq_id = Some(i), "_atom_site.auth_seq_id" => cols.auth_seq_id = Some(i), + "_atom_site.label_entity_id" => cols.label_entity_id = Some(i), "_atom_site.pdbx_PDB_ins_code" => cols.pdbx_pdb_ins_code = Some(i), "_atom_site.label_alt_id" => cols.label_alt_id = Some(i), "_atom_site.B_iso_or_equiv" => cols.b_iso_or_equiv = Some(i), @@ -512,6 +514,11 @@ fn parse_atom_site_loop( .and_then(|s| s.chars().next()) .unwrap_or(' '); + // mmCIF label residue ids, kept alongside the auth ids above + let label_seq_id = col!(cols.label_seq_id) + .and_then(|s| s.parse().ok()) + .filter(|&seq_id| seq_id > 0); + let alt = col!(cols.label_alt_id) .and_then(|s| s.chars().next()) .unwrap_or(' '); @@ -551,6 +558,12 @@ fn parse_atom_site_loop( model .source_chains .push(col!(cols.label_asym_id).unwrap_or("")); + let labels = ParsedLabels::new( + col!(cols.label_asym_id), + col!(cols.label_entity_id), + label_seq_id, + model.atoms.last(), + ); model.atoms.push(ParsedAtom { name: atom_name.to_string(), element, @@ -565,6 +578,7 @@ fn parse_atom_site_loop( occupancy, b_factor, segi: String::new(), + labels, }); model.coords.push(Vec3::new(x, y, z)); } @@ -798,6 +812,7 @@ fn parse_ss_single( mod tests { use super::*; use patinae_mol::SecondaryStructure; + use std::sync::Arc; #[test] fn test_read_simple_cif() { @@ -1172,4 +1187,73 @@ _atom_site.pdbx_PDB_model_num assert_eq!(molecule.state_count(), 1); assert_eq!(molecule.atoms_slice()[0].residue.chain, "A"); } + + #[test] + fn label_fields_are_kept_per_atom_when_they_differ_from_auth() { + let cif_data = r#"data_AF3 +loop_ +_atom_site.group_PDB +_atom_site.id +_atom_site.type_symbol +_atom_site.label_atom_id +_atom_site.label_comp_id +_atom_site.label_asym_id +_atom_site.label_entity_id +_atom_site.label_seq_id +_atom_site.auth_atom_id +_atom_site.auth_comp_id +_atom_site.auth_asym_id +_atom_site.auth_seq_id +_atom_site.Cartn_x +_atom_site.Cartn_y +_atom_site.Cartn_z +ATOM 1 N N MSE B 2 1 N MET H 101 0.0 0.0 0.0 +ATOM 2 SE SE MSE B 2 1 SE MET H 101 1.5 0.0 0.0 +ATOM 3 N N GLY B 2 2 N GLY H 102 3.0 0.0 0.0 +HETATM 4 O O HOH D 4 . O HOH H 201 6.0 0.0 0.0 +"#; + + let mol = CifReader::new(cif_data.as_bytes()).read().unwrap(); + let atoms = mol.atoms_slice(); + + let first = &atoms[0].residue; + assert_eq!((first.chain.as_str(), first.resv), ("H", 101)); + assert_eq!(first.resn, "MET"); + assert_eq!(first.label_asym_id.as_deref(), Some("B")); + assert_eq!(first.label_entity_id.as_deref(), Some("2")); + assert_eq!(first.label_seq_id, Some(1)); + assert!(Arc::ptr_eq(&atoms[0].residue, &atoms[1].residue)); + + assert_eq!(atoms[2].residue.label_seq_id, Some(2)); + assert!(!Arc::ptr_eq(&atoms[1].residue, &atoms[2].residue)); + + let water = &atoms[3].residue; + assert_eq!(water.label_asym_id.as_deref(), Some("D")); + assert_eq!(water.label_entity_id.as_deref(), Some("4")); + assert_eq!(water.label_seq_id, None); + } + + #[test] + fn missing_label_columns_leave_label_fields_empty() { + let cif_data = r#"data_AUTH +loop_ +_atom_site.id +_atom_site.type_symbol +_atom_site.auth_atom_id +_atom_site.auth_comp_id +_atom_site.auth_asym_id +_atom_site.auth_seq_id +_atom_site.Cartn_x +_atom_site.Cartn_y +_atom_site.Cartn_z +1 N N ALA A 1 0.0 0.0 0.0 +"#; + + let mol = CifReader::new(cif_data.as_bytes()).read().unwrap(); + let atom = &mol.atoms_slice()[0]; + + assert_eq!(atom.residue.label_asym_id, None); + assert_eq!(atom.residue.label_entity_id, None); + assert_eq!(atom.residue.label_seq_id, None); + } } diff --git a/crates/patinae-io/src/logical_models.rs b/crates/patinae-io/src/logical_models.rs index 2af849b..10f9f97 100644 --- a/crates/patinae-io/src/logical_models.rs +++ b/crates/patinae-io/src/logical_models.rs @@ -22,17 +22,22 @@ pub(crate) struct ParsedAtom { pub(crate) occupancy: f32, pub(crate) b_factor: f32, pub(crate) segi: String, + pub(crate) labels: ParsedLabels, } impl ParsedAtom { fn residue(&self) -> AtomResidue { - AtomResidue::from_parts( + let mut residue = AtomResidue::from_parts( self.chain.clone(), self.resn.clone(), self.resv, self.icode, self.segi.clone(), - ) + ); + residue.label_asym_id = self.labels.asym_id.as_deref().map(str::to_owned); + residue.label_entity_id = self.labels.entity_id.as_deref().map(str::to_owned); + residue.label_seq_id = self.labels.seq_id; + residue } /// Whether [`ParsedAtom::residue`] gives equal residues for both atoms. @@ -42,6 +47,43 @@ impl ParsedAtom { && self.resv == other.resv && self.icode == other.icode && self.segi == other.segi + && self.labels == other.labels + } +} + +/// mmCIF `label_*` identifiers, kept alongside the `auth_*` values used for display. +#[derive(Debug, Clone, Default, PartialEq, Eq, Hash)] +pub(crate) struct ParsedLabels { + pub(crate) asym_id: Option>, + pub(crate) entity_id: Option>, + pub(crate) seq_id: Option, +} + +impl ParsedLabels { + /// Atoms of one residue are contiguous, so ids equal to the previous atom's share its strings. + pub(crate) fn new( + asym_id: Option<&str>, + entity_id: Option<&str>, + seq_id: Option, + previous: Option<&ParsedAtom>, + ) -> Self { + let previous = previous.map(|atom| &atom.labels); + Self { + asym_id: shared(asym_id, previous.and_then(|labels| labels.asym_id.as_ref())), + entity_id: shared( + entity_id, + previous.and_then(|labels| labels.entity_id.as_ref()), + ), + seq_id, + } + } +} + +fn shared(value: Option<&str>, previous: Option<&Arc>) -> Option> { + let value = value?; + match previous { + Some(previous) if &**previous == value => Some(previous.clone()), + _ => Some(value.into()), } } @@ -102,6 +144,7 @@ impl SourceChains { } // File-level atom serials are metadata and may continue across compatible models. +// mmCIF label ids are metadata too: grouped models keep the first model's labels. #[derive(Debug, Clone, PartialEq, Eq, Hash)] struct AtomIdentity { name: String, @@ -298,6 +341,7 @@ mod tests { occupancy: 1.0, b_factor: 0.0, segi: String::new(), + labels: ParsedLabels::default(), }); model.coords.push(Vec3::new(1.0, 2.0, 3.0)); model @@ -334,9 +378,10 @@ mod tests { fn residues_are_shared_by_value_not_by_adjacency() { let mut model = model(); let template = model.atoms[0].clone(); - for resv in [1, 2, 1] { + for (resv, seq_id) in [(1, None), (1, Some(1)), (2, None), (1, None)] { model.atoms.push(ParsedAtom { resv, + labels: ParsedLabels::new(None, None, seq_id, model.atoms.last()), ..template.clone() }); model.coords.push(Vec3::new(0.0, 0.0, 0.0)); @@ -346,9 +391,32 @@ mod tests { let residues: Vec<_> = molecule.atoms().map(|atom| &atom.residue).collect(); assert!(Arc::ptr_eq(residues[0], residues[1])); + // Same auth ids, different label_seq_id: a distinct residue. assert!(!Arc::ptr_eq(residues[1], residues[2])); + assert_eq!(residues[2].label_seq_id, Some(1)); // Non-contiguous atoms of one residue still share it. - assert!(Arc::ptr_eq(residues[0], residues[3])); - assert!(!Arc::ptr_eq(residues[2], residues[3])); + assert!(Arc::ptr_eq(residues[0], residues[4])); + assert!(!Arc::ptr_eq(residues[3], residues[4])); + } + + #[test] + fn equal_labels_share_the_previous_atom_strings() { + let mut model = model(); + let template = model.atoms[0].clone(); + for asym_id in ["B", "B", "C"] { + model.atoms.push(ParsedAtom { + labels: ParsedLabels::new(Some(asym_id), Some("1"), None, model.atoms.last()), + ..template.clone() + }); + } + let labels: Vec<_> = model.atoms[1..].iter().map(|atom| &atom.labels).collect(); + + let asym = |index: usize| labels[index].asym_id.as_ref().unwrap(); + assert!(Arc::ptr_eq(asym(0), asym(1))); + assert_eq!(&**asym(2), "C"); + assert!(Arc::ptr_eq( + labels[0].entity_id.as_ref().unwrap(), + labels[2].entity_id.as_ref().unwrap() + )); } } diff --git a/crates/patinae-io/src/pdb/parser.rs b/crates/patinae-io/src/pdb/parser.rs index 42f2507..28f2f1e 100644 --- a/crates/patinae-io/src/pdb/parser.rs +++ b/crates/patinae-io/src/pdb/parser.rs @@ -10,7 +10,7 @@ use patinae_mol::{AtomIndex, BondOrder, ObjectMolecule, SecondaryStructure}; use crate::assembly::PdbAssemblies; use crate::error::{IoError, IoResult}; -use crate::logical_models::{build_molecules, ParsedAtom, ParsedModel}; +use crate::logical_models::{build_molecules, ParsedAtom, ParsedLabels, ParsedModel}; use crate::pdb::hybrid36::hy36decode; use crate::traits::MoleculeReader; @@ -250,6 +250,7 @@ fn parsed_atom_from_record(record: &AtomRecord, effective_chain: String) -> Pars occupancy: record.occupancy, b_factor: record.b_factor, segi: record.segi.clone(), + labels: ParsedLabels::default(), } } @@ -904,4 +905,17 @@ ENDMDL assert_eq!(molecule.state_count(), 1); assert_eq!(molecule.atoms_slice()[0].residue.chain, "A"); } + + #[test] + fn pdb_atoms_have_no_mmcif_label_fields() { + let pdb = + "ATOM 1 CA ALA A 1 0.000 0.000 0.000 1.00 20.00 C\n"; + + let molecule = crate::pdb::read_pdb_str(pdb).unwrap(); + let atom = &molecule.atoms_slice()[0]; + + assert_eq!(atom.residue.label_asym_id, None); + assert_eq!(atom.residue.label_entity_id, None); + assert_eq!(atom.residue.label_seq_id, None); + } } diff --git a/crates/patinae-mm/src/mutate.rs b/crates/patinae-mm/src/mutate.rs index 14438b1..80c8a7b 100644 --- a/crates/patinae-mm/src/mutate.rs +++ b/crates/patinae-mm/src/mutate.rs @@ -8,7 +8,7 @@ use std::collections::HashMap; use std::sync::Arc; -use patinae_mol::{AtomIndex, AtomResidue, BondOrder, ObjectMolecule}; +use patinae_mol::{AtomIndex, BondOrder, ObjectMolecule}; use crate::residue_geometry::ideal_side_chain_on; @@ -69,13 +69,10 @@ pub fn build_mutant( let side_chain = ideal_side_chain_on(target_resn, n, ca, c) .ok_or_else(|| format!("residue {target_resn} is not supported"))?; - let residue = Arc::new(AtomResidue::from_parts( - chain, - target_resn, - resv, - inscode, - template.residue.segi.clone(), - )); + // Clone the template residue so file-level ids (segi, mmCIF labels) survive. + let mut residue = (*template.residue).clone(); + residue.key.resn = target_resn.to_owned(); + let residue = Arc::new(residue); let mut name_to_idx: HashMap = HashMap::new(); for idx in &survivors { if let Some(atom) = mol.get_atom_mut(*idx) { @@ -103,3 +100,46 @@ pub fn build_mutant( mol.regroup_by_residue(); Ok(mol) } + +#[cfg(test)] +mod tests { + use super::*; + use lin_alg::f32::Vec3; + use patinae_mol::{Atom, AtomResidue, CoordSet, Element}; + + #[test] + fn mutation_keeps_file_level_residue_ids() { + let mut residue = AtomResidue::from_parts("H", "GLY", 101, ' ', "SEG"); + residue.label_asym_id = Some("B".to_owned()); + residue.label_entity_id = Some("2".to_owned()); + residue.label_seq_id = Some(1); + let residue = Arc::new(residue); + + let mut source = ObjectMolecule::new("gly"); + for (name, element) in [ + ("N", Element::Nitrogen), + ("CA", Element::Carbon), + ("C", Element::Carbon), + ] { + let mut atom = Atom::new(name, element); + atom.residue = residue.clone(); + source.add_atom(atom); + } + source.add_coord_set(CoordSet::from_vec3(&[ + Vec3::new(-0.5, 1.4, 0.0), + Vec3::new(0.0, 0.0, 0.0), + Vec3::new(1.5, 0.0, 0.0), + ])); + + let mutant = build_mutant(&source, "H", 101, ' ', "ALA").unwrap(); + + assert!(mutant.atoms().any(|atom| &*atom.name == "CB")); + for atom in mutant.atoms() { + assert_eq!(atom.residue.resn, "ALA"); + assert_eq!(atom.residue.segi, "SEG"); + assert_eq!(atom.residue.label_asym_id.as_deref(), Some("B")); + assert_eq!(atom.residue.label_entity_id.as_deref(), Some("2")); + assert_eq!(atom.residue.label_seq_id, Some(1)); + } + } +} diff --git a/crates/patinae-mol/src/atom.rs b/crates/patinae-mol/src/atom.rs index 0c42cfa..c4af1b8 100644 --- a/crates/patinae-mol/src/atom.rs +++ b/crates/patinae-mol/src/atom.rs @@ -329,6 +329,17 @@ pub struct AtomResidue { pub key: ResidueKey, /// Segment identifier pub segi: String, + /// mmCIF `label_asym_id` (`None` for formats without the label scheme). + /// + /// Label ids are kept as read from the file; edits do not renumber them. + #[serde(default)] + pub label_asym_id: Option, + /// mmCIF `label_entity_id` + #[serde(default)] + pub label_entity_id: Option, + /// mmCIF `label_seq_id`, the 1-based `_entity_poly_seq` position (`None` for non-polymers) + #[serde(default)] + pub label_seq_id: Option, } impl Deref for AtomResidue { @@ -341,17 +352,20 @@ impl Deref for AtomResidue { impl Default for AtomResidue { fn default() -> Self { - AtomResidue { - key: ResidueKey::new("", "", 0, ' '), - segi: String::new(), - } + AtomResidue::new(ResidueKey::new("", "", 0, ' '), String::new()) } } impl AtomResidue { /// Create a new AtomResidue pub fn new(key: ResidueKey, segi: String) -> Self { - AtomResidue { key, segi } + AtomResidue { + key, + segi, + label_asym_id: None, + label_entity_id: None, + label_seq_id: None, + } } /// Create a new AtomResidue from individual fields @@ -362,10 +376,7 @@ impl AtomResidue { inscode: char, segi: impl Into, ) -> Self { - AtomResidue { - key: ResidueKey::new(chain, resn, resv, inscode), - segi: segi.into(), - } + AtomResidue::new(ResidueKey::new(chain, resn, resv, inscode), segi.into()) } } @@ -959,4 +970,24 @@ mod tests { // But ss_type is per-atom assert_eq!(atom1.ss_type, SecondaryStructure::Helix); } + + #[test] + fn residue_without_label_fields_deserializes_from_older_data() { + #[derive(Serialize)] + struct LegacyAtomResidue { + key: ResidueKey, + segi: String, + } + + let legacy = LegacyAtomResidue { + key: ResidueKey::new("A", "ALA", 1, ' '), + segi: String::new(), + }; + let bytes = rmp_serde::to_vec_named(&legacy).unwrap(); + let residue: AtomResidue = rmp_serde::from_slice(&bytes).unwrap(); + + assert_eq!(residue, AtomResidue::from_parts("A", "ALA", 1, ' ', "")); + assert_eq!(residue.label_asym_id, None); + assert_eq!(residue.label_seq_id, None); + } } From 9a572833eca1fa15f18ae55377d8de36aa982432 Mon Sep 17 00:00:00 2001 From: norsage Date: Sat, 26 Sep 2026 12:59:31 +0300 Subject: [PATCH 4/5] feat(io): read mmCIF entity and polymer sequence tables Parse _entity, _entity_poly and _entity_poly_seq from mmCIF and bCIF into ObjectMolecule::entities. The full polymer sequence includes unresolved residues and is indexed by num - 1, so label_seq_id on a residue points straight into it. - Rows may come in any order, but num must run 1..N without gaps, as the PDBx/mmCIF dictionary requires; a skipped num is a parse error. - num must be at most 1_000_000, which keeps a corrupt value from allocating a huge sequence. - Microheterogeneity keeps the first monomer in the sequence and the rest in EntityPolymer::alternatives. - Entity errors are reported only for blocks with atoms, as for assemblies. Entity and polymer types map the PDBx/mmCIF enumerations and keep unknown values verbatim. Other formats leave entities empty. Entities describe the file as loaded; edits do not update them. Sessions store the new field, so the PRS format version is 5. Version 4 files without entities still load. With label ids and entity tables, reading 487 PDB mmCIF files costs 80.7G instructions retired, against 86.5G before this series. Co-Authored-By: Claude Opus 5.5 --- crates/patinae-io/src/bcif/parser.rs | 56 +++++ crates/patinae-io/src/cif/parser.rs | 77 ++++++ crates/patinae-io/src/entity.rs | 340 +++++++++++++++++++++++++++ crates/patinae-io/src/lib.rs | 1 + crates/patinae-mol/src/entity.rs | 134 +++++++++++ crates/patinae-mol/src/lib.rs | 2 + crates/patinae-mol/src/molecule.rs | 6 + crates/patinae-session/src/prs.rs | 60 ++++- 8 files changed, 672 insertions(+), 4 deletions(-) create mode 100644 crates/patinae-io/src/entity.rs create mode 100644 crates/patinae-mol/src/entity.rs diff --git a/crates/patinae-io/src/bcif/parser.rs b/crates/patinae-io/src/bcif/parser.rs index c020360..ac115bc 100644 --- a/crates/patinae-io/src/bcif/parser.rs +++ b/crates/patinae-io/src/bcif/parser.rs @@ -11,6 +11,7 @@ use patinae_mol::{Element, ObjectMolecule, SecondaryStructure}; use crate::assembly::{AssemblyRow, AssemblyRows, CATEGORIES}; use crate::cif::common::{apply_secondary_structure, SecondaryStructureRange, SsCategory}; +use crate::entity::{EntityCategory, EntityRows}; use crate::error::{IoError, IoResult}; use crate::logical_models::{build_molecules, ParsedAtom, ParsedLabels, ParsedModel}; use crate::traits::MoleculeReader; @@ -193,11 +194,15 @@ fn parse_data_block(block: BcifDataBlock, bond_tolerance: f32) -> IoResult parse_atom_site(category, &mut models)?, "_cell" => parse_cell(category, &mut cell)?, @@ -216,6 +221,7 @@ fn parse_data_block(block: BcifDataBlock, bond_tolerance: f32) -> IoResult IoResult IoResult<()> { + let cols = + CategoryColumns::decode_selected(category, &["id", "type", "entity_id", "num", "mon_id"])?; + for i in 0..category.row_count as usize { + entities.push(entity_category, |name| { + cols.code_at(name, i).map(Cow::into_owned) + }); + } + Ok(()) +} + fn parse_cell(category: &BcifCategory, cell: &mut CellInfo) -> IoResult<()> { let cols = CategoryColumns::decode_selected( category, @@ -662,4 +684,38 @@ mod tests { assert_eq!(residue.label_entity_id.as_deref(), Some("2")); assert_eq!(residue.label_seq_id, Some(1)); } + + #[test] + fn entity_tables_are_read_from_typed_columns() { + let mut block = atom_site_block("ENT", &[atom_row(1, "CA", "GLY", "A", 0.0, 1)]); + block.categories.extend([ + BcifCategory { + name: "_entity".to_string(), + row_count: 1, + columns: vec![ + int_column("id", [1].into_iter()), + string_column("type", ["polymer"].into_iter()), + ], + }, + BcifCategory { + name: "_entity_poly_seq".to_string(), + row_count: 2, + columns: vec![ + string_column("entity_id", ["1", "1"].into_iter()), + int_column("num", [1, 2].into_iter()), + string_column("mon_id", ["MET", "GLY"].into_iter()), + ], + }, + ]); + + let molecule = + crate::bcif::read_bcif_bytes_with_bond_tolerance(&encode_bcif_file(&[block]), 0.1) + .unwrap(); + + assert_eq!(molecule.entities.len(), 1); + let entity = &molecule.entities[0]; + assert_eq!(entity.id, "1"); + assert_eq!(entity.kind, Some(patinae_mol::EntityKind::Polymer)); + assert_eq!(entity.polymer.as_ref().unwrap().sequence, ["MET", "GLY"]); + } } diff --git a/crates/patinae-io/src/cif/parser.rs b/crates/patinae-io/src/cif/parser.rs index 7c61ed6..2c3a6f3 100644 --- a/crates/patinae-io/src/cif/parser.rs +++ b/crates/patinae-io/src/cif/parser.rs @@ -9,6 +9,7 @@ use lin_alg::f32::Vec3; use patinae_mol::{Element, ObjectMolecule}; use crate::assembly::{AssemblyRow, AssemblyRows, CATEGORIES}; +use crate::entity::{EntityCategory, EntityRows}; use crate::error::{IoError, IoResult}; use crate::logical_models::{build_molecules, ParsedAtom, ParsedLabels, ParsedModel}; use crate::traits::MoleculeReader; @@ -54,6 +55,7 @@ struct CifBlockData { assemblies: AssemblyRows, assembly_singles: BTreeMap, assembly_error: Option, + entities: EntityRows, } impl CifBlockData { @@ -68,6 +70,7 @@ impl CifBlockData { assemblies: AssemblyRows::default(), assembly_singles: BTreeMap::new(), assembly_error: None, + entities: EntityRows::default(), } } } @@ -277,6 +280,9 @@ fn parse_cif_block( &mut block.ss_ranges, ); } + Token::DataName(name) if EntityCategory::of_data_name(name).is_some() => { + pos = parse_entity_single(tokens, pos, &mut block.entities); + } Token::DataName(name) => { if let Some((category, field)) = name.split_once('.') { if CATEGORIES.contains(&category) { @@ -309,6 +315,7 @@ fn parse_cif_block( block.assemblies.push(&category, row); } let definitions = block.assemblies.resolve()?; + let entities = block.entities.resolve()?; if definitions.is_empty() { for model in block.models.values_mut() { model.source_chains.clear(); @@ -319,6 +326,7 @@ fn parse_cif_block( for mol in &mut molecules { mol.assembly.definitions = definitions.clone(); + mol.entities = entities.clone(); block.cell.apply_to(mol, block.space_group.as_deref()); apply_secondary_structure(mol, &block.ss_ranges); @@ -360,6 +368,9 @@ fn parse_loop(tokens: &[Token], mut pos: usize, block: &mut CifBlockData) -> IoR .and_then(|name| name.split_once('.')) .map(|(category, _)| category) .filter(|category| CATEGORIES.contains(category)); + let entity_category = columns + .first() + .and_then(|name| EntityCategory::of_data_name(name)); let data_start = pos; // Check loop category @@ -407,6 +418,14 @@ fn parse_loop(tokens: &[Token], mut pos: usize, block: &mut CifBlockData) -> IoR block.assembly_error.get_or_insert(error); } } + if let Some(category) = entity_category { + parse_entity_loop( + &tokens[data_start..pos], + &columns, + category, + &mut block.entities, + ); + } Ok(pos) } @@ -610,6 +629,64 @@ fn parse_assembly_loop( Ok(()) } +/// Collect entity rows from a loop already located by the block parser. +fn parse_entity_loop( + tokens: &[Token], + columns: &[&str], + category: EntityCategory, + entities: &mut EntityRows, +) { + let fields: Vec> = columns + .iter() + .map(|name| name.split_once('.').map(|(_, field)| field)) + .collect(); + let mut chunks = tokens.chunks_exact(columns.len()); + for row in &mut chunks { + entities.push(category, |name| { + let index = fields.iter().position(|field| *field == Some(name))?; + token_value_str(&row[index]).map(str::to_owned) + }); + } + if !chunks.remainder().is_empty() { + entities.fail(IoError::parse_msg("Incomplete entity loop row")); + } +} + +/// Parse consecutive single-item data names of one entity category +fn parse_entity_single(tokens: &[Token], mut pos: usize, entities: &mut EntityRows) -> usize { + let Some(Token::DataName(first)) = tokens.get(pos) else { + return pos + 1; + }; + let Some((prefix, _)) = first.split_once('.') else { + return pos + 1; + }; + let Some(category) = EntityCategory::from_name(prefix) else { + return pos + 1; + }; + + let mut fields: HashMap<&str, &str> = HashMap::new(); + while let Some(Token::DataName(name)) = tokens.get(pos) { + let Some((item_prefix, field)) = name.split_once('.') else { + break; + }; + if item_prefix != prefix { + break; + } + pos += 1; + if let Some(value) = tokens.get(pos).and_then(token_value_str) { + fields.insert(field, value); + pos += 1; + } else if matches!(tokens.get(pos), Some(Token::Missing | Token::Unknown)) { + pos += 1; + } + } + + entities.push(category, |name| { + fields.get(name).map(|value| value.to_string()) + }); + pos +} + /// Parse _cell data items fn parse_cell(tokens: &[Token], mut pos: usize, cell: &mut CellInfo) -> usize { while pos < tokens.len() { diff --git a/crates/patinae-io/src/entity.rs b/crates/patinae-io/src/entity.rs new file mode 100644 index 0000000..0b8f331 --- /dev/null +++ b/crates/patinae-io/src/entity.rs @@ -0,0 +1,340 @@ +//! Shared mmCIF entity decoding (`_entity`, `_entity_poly`, `_entity_poly_seq`). + +use patinae_mol::{Entity, EntityKind, EntityPolymer, PolymerKind}; + +use crate::error::{IoError, IoResult}; + +/// Keeps a corrupt `num` from allocating a huge dense sequence (titin is ~35 000 residues). +const MAX_ENTITY_POLY_SEQ_NUM: u32 = 1_000_000; + +/// Entity categories read from mmCIF and bCIF. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub(crate) enum EntityCategory { + Entity, + Poly, + PolySeq, +} + +impl EntityCategory { + pub(crate) fn from_name(category: &str) -> Option { + match category { + "_entity" => Some(Self::Entity), + "_entity_poly" => Some(Self::Poly), + "_entity_poly_seq" => Some(Self::PolySeq), + _ => None, + } + } + + /// Category of an mmCIF data name such as `_entity_poly.type`. + pub(crate) fn of_data_name(name: &str) -> Option { + name.split_once('.') + .and_then(|(category, _)| Self::from_name(category)) + } +} + +/// `_entity` / `_entity_poly` row: an entity id and its type. +struct KindRow { + id: Option, + kind: Option, +} + +/// `_entity_poly_seq` row. +struct MonomerRow { + entity_id: Option, + num: Option, + mon_id: Option, +} + +/// Rows are validated in [`EntityRows::resolve`], so metadata-only blocks still load. +#[derive(Default)] +pub(crate) struct EntityRows { + entities: Vec, + polymers: Vec, + monomers: Vec, + error: Option, +} + +fn required<'a>(value: &'a Option, field: &str) -> IoResult<&'a str> { + value + .as_deref() + .ok_or_else(|| IoError::parse_msg(format!("Missing field '{field}'"))) +} + +impl EntityRows { + /// Add one row from a field lookup keyed by the name after the category prefix. + pub(crate) fn push( + &mut self, + category: EntityCategory, + field: impl Fn(&str) -> Option, + ) { + match category { + EntityCategory::Entity => self.entities.push(KindRow { + id: field("id"), + kind: field("type"), + }), + EntityCategory::Poly => self.polymers.push(KindRow { + id: field("entity_id"), + kind: field("type"), + }), + EntityCategory::PolySeq => self.monomers.push(MonomerRow { + entity_id: field("entity_id"), + num: field("num"), + mon_id: field("mon_id"), + }), + } + } + + /// Record a malformed category; reported by [`EntityRows::resolve`]. + pub(crate) fn fail(&mut self, error: IoError) { + self.error.get_or_insert(error); + } + + /// Entities in `_entity` order, followed by ids that only the polymer tables mention. + pub(crate) fn resolve(self) -> IoResult> { + if let Some(error) = self.error { + return Err(error); + } + let mut entities: Vec = Vec::new(); + for row in &self.entities { + let id = required(&row.id, "_entity.id")?; + if entities.iter().any(|entity| entity.id == id) { + return Err(IoError::parse_msg(format!("Duplicate entity '{id}'"))); + } + entities.push(Entity { + id: id.to_owned(), + kind: row.kind.as_deref().map(EntityKind::from_mmcif), + polymer: None, + }); + } + + for row in &self.polymers { + let id = required(&row.id, "_entity_poly.entity_id")?; + polymer_mut(&mut entities, id).kind = row.kind.as_deref().map(PolymerKind::from_mmcif); + } + + for row in &self.monomers { + let id = required(&row.entity_id, "_entity_poly_seq.entity_id")?; + let num = required(&row.num, "_entity_poly_seq.num")?; + let mon_id = required(&row.mon_id, "_entity_poly_seq.mon_id")?; + let num = num + .parse::() + .ok() + .filter(|num| (1..=MAX_ENTITY_POLY_SEQ_NUM).contains(num)) + .ok_or_else(|| { + IoError::parse_msg(format!("Invalid _entity_poly_seq.num '{num}'")) + })?; + + let polymer = polymer_mut(&mut entities, id); + let index = num as usize - 1; + if polymer.sequence.len() <= index { + polymer.sequence.resize(index + 1, String::new()); + } + // Microheterogeneity: the first monomer is the primary one. + if polymer.sequence[index].is_empty() { + polymer.sequence[index] = mon_id.to_owned(); + } else if polymer.sequence[index] != mon_id { + let alternatives = polymer.alternatives.entry(num).or_default(); + if !alternatives.iter().any(|alternative| alternative == mon_id) { + alternatives.push(mon_id.to_owned()); + } + } + } + + // The dictionary numbers monomers 1..N; a skipped `num` means rows were lost. + for entity in &entities { + let Some(polymer) = &entity.polymer else { + continue; + }; + if let Some(index) = polymer.sequence.iter().position(String::is_empty) { + return Err(IoError::parse_msg(format!( + "Entity '{}' skips _entity_poly_seq.num {}", + entity.id, + index + 1 + ))); + } + } + + Ok(entities) + } +} + +fn polymer_mut<'a>(entities: &'a mut Vec, id: &str) -> &'a mut EntityPolymer { + let index = match entities.iter().position(|entity| entity.id == id) { + Some(index) => index, + None => { + entities.push(Entity { + id: id.to_owned(), + kind: None, + polymer: None, + }); + entities.len() - 1 + } + }; + entities[index].polymer.get_or_insert_with(Default::default) +} + +#[cfg(test)] +mod tests { + use patinae_mol::ObjectMolecule; + + use super::*; + + const ATOMS: &str = "loop_ +_atom_site.id +_atom_site.type_symbol +_atom_site.label_atom_id +_atom_site.label_comp_id +_atom_site.label_asym_id +_atom_site.label_entity_id +_atom_site.label_seq_id +_atom_site.Cartn_x +_atom_site.Cartn_y +_atom_site.Cartn_z +1 C CA GLY A 1 2 0.0 0.0 0.0 +2 O O HOH B 2 . 5.0 0.0 0.0 +"; + + fn cif(text: &str) -> IoResult { + crate::cif::read_cif_str(&format!("data_test\n{text}{ATOMS}")) + } + + fn polymer(mol: &ObjectMolecule, id: &str) -> EntityPolymer { + let entity = mol.entities.iter().find(|e| e.id == id).unwrap(); + entity.polymer.clone().unwrap() + } + + #[test] + fn entities_keep_kinds_and_full_sequence_with_unresolved_residues() { + let mol = cif("loop_ +_entity.id +_entity.type +1 polymer +2 water +_entity_poly.entity_id 1 +_entity_poly.type 'polypeptide(L)' +loop_ +_entity_poly_seq.entity_id +_entity_poly_seq.num +_entity_poly_seq.mon_id +1 1 MET +1 2 GLY +1 3 SER +") + .unwrap(); + + assert_eq!(mol.entities.len(), 2); + assert_eq!(mol.entities[0].kind, Some(EntityKind::Polymer)); + assert_eq!(mol.entities[1].kind, Some(EntityKind::Water)); + assert_eq!(mol.entities[1].polymer, None); + + let chain = polymer(&mol, "1"); + assert_eq!(chain.kind, Some(PolymerKind::PeptideL)); + assert_eq!(chain.sequence, ["MET", "GLY", "SER"]); + let atom = &mol.atoms_slice()[0]; + assert_eq!(atom.residue.label_entity_id.as_deref(), Some("1")); + assert_eq!( + chain.monomer(atom.residue.label_seq_id.unwrap()), + Some("GLY") + ); + } + + #[test] + fn unsorted_and_microheterogeneous_sequences_keep_positions() { + let mol = cif("loop_ +_entity_poly_seq.entity_id +_entity_poly_seq.num +_entity_poly_seq.mon_id +_entity_poly_seq.hetero +1 3 SER n +1 1 MET n +1 2 GLY y +1 2 ALA y +1 2 GLY y +1 4 LYS n +") + .unwrap(); + + let chain = polymer(&mol, "1"); + assert_eq!(chain.kind, None); + assert_eq!(chain.sequence, ["MET", "GLY", "SER", "LYS"]); + assert_eq!(chain.alternatives.len(), 1); + assert_eq!(chain.alternatives[&2], ["ALA"]); + assert_eq!(chain.monomer(4), Some("LYS")); + assert_eq!(mol.entities[0].kind, None); + } + + #[test] + fn single_item_entity_categories_are_read() { + let mol = cif("_entity.id 1 +_entity.type polymer +_entity_poly.entity_id 1 +_entity_poly.type polyribonucleotide +_entity_poly_seq.entity_id 1 +_entity_poly_seq.num 1 +_entity_poly_seq.mon_id G +") + .unwrap(); + + let chain = polymer(&mol, "1"); + assert_eq!(chain.kind, Some(PolymerKind::Rna)); + assert_eq!(chain.sequence, ["G"]); + } + + #[test] + fn corrupt_sequence_numbers_are_rejected() { + for num in ["0", "-1", "x", "1000001"] { + let text = format!( + "_entity_poly_seq.entity_id 1\n_entity_poly_seq.num {num}\n_entity_poly_seq.mon_id MET\n" + ); + let error = cif(&text).err().unwrap(); + assert!( + error.to_string().contains("Invalid _entity_poly_seq.num"), + "{num}" + ); + } + } + + #[test] + fn skipped_sequence_numbers_are_rejected() { + let error = cif("loop_ +_entity_poly_seq.entity_id +_entity_poly_seq.num +_entity_poly_seq.mon_id +1 1 MET +1 3 SER +") + .err() + .unwrap(); + + assert!(error + .to_string() + .contains("Entity '1' skips _entity_poly_seq.num 2")); + } + + #[test] + fn duplicate_entities_and_missing_ids_are_rejected() { + let duplicate = cif("loop_\n_entity.id\n_entity.type\n1 polymer\n1 water\n"); + assert!(duplicate + .err() + .unwrap() + .to_string() + .contains("Duplicate entity")); + + let missing = cif("_entity_poly_seq.entity_id 1\n_entity_poly_seq.num 1\n"); + assert!(missing + .err() + .unwrap() + .to_string() + .contains("_entity_poly_seq.mon_id")); + } + + #[test] + fn formats_without_entities_leave_them_empty() { + let mol = crate::cif::read_cif_str(&format!("data_test\n{ATOMS}")).unwrap(); + assert!(mol.entities.is_empty()); + + let pdb = + "ATOM 1 CA GLY A 1 0.000 0.000 0.000 1.00 0.00 C\n"; + assert!(crate::pdb::read_pdb_str(pdb).unwrap().entities.is_empty()); + } +} diff --git a/crates/patinae-io/src/lib.rs b/crates/patinae-io/src/lib.rs index 1912d79..a0bddd3 100644 --- a/crates/patinae-io/src/lib.rs +++ b/crates/patinae-io/src/lib.rs @@ -60,6 +60,7 @@ pub mod ccp4; pub mod cif; pub mod compress; pub mod detect; +mod entity; pub mod error; pub mod gro; mod logical_models; diff --git a/crates/patinae-mol/src/entity.rs b/crates/patinae-mol/src/entity.rs new file mode 100644 index 0000000..397e10e --- /dev/null +++ b/crates/patinae-mol/src/entity.rs @@ -0,0 +1,134 @@ +//! mmCIF entities, referenced from atoms by `label_entity_id` and `label_seq_id`. +//! +//! Entities describe the structure as loaded from the file. Edits made afterwards +//! (mutation, atom removal) do not update them, so an edited residue may differ from +//! [`EntityPolymer::monomer`] at its `label_seq_id`. + +use std::collections::BTreeMap; + +use serde::{Deserialize, Serialize}; + +/// A chemically distinct part of the structure (`_entity`). +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +pub struct Entity { + /// `_entity.id`, matched by `AtomResidue::label_entity_id` + pub id: String, + /// `_entity.type`; `None` when the file does not state it + pub kind: Option, + /// Polymer data from `_entity_poly` and `_entity_poly_seq` + pub polymer: Option, +} + +/// `_entity.type` +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +pub enum EntityKind { + Polymer, + NonPolymer, + Branched, + Macrolide, + Water, + /// Value outside the PDBx/mmCIF enumeration, kept verbatim + Other(String), +} + +impl EntityKind { + /// Parse an `_entity.type` value (case-insensitive). + pub fn from_mmcif(value: &str) -> Self { + match value.to_ascii_lowercase().as_str() { + "polymer" => Self::Polymer, + "non-polymer" => Self::NonPolymer, + "branched" => Self::Branched, + "macrolide" => Self::Macrolide, + "water" => Self::Water, + _ => Self::Other(value.to_owned()), + } + } +} + +/// `_entity_poly.type` +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +pub enum PolymerKind { + PeptideL, + PeptideD, + Dna, + Rna, + DnaRnaHybrid, + PeptideNucleicAcid, + CyclicPseudoPeptide, + /// Value outside the PDBx/mmCIF enumeration (including `other`), kept verbatim + Other(String), +} + +impl PolymerKind { + /// Parse an `_entity_poly.type` value (case-insensitive). + pub fn from_mmcif(value: &str) -> Self { + match value.to_ascii_lowercase().as_str() { + "polypeptide(l)" => Self::PeptideL, + "polypeptide(d)" => Self::PeptideD, + "polydeoxyribonucleotide" => Self::Dna, + "polyribonucleotide" => Self::Rna, + "polydeoxyribonucleotide/polyribonucleotide hybrid" => Self::DnaRnaHybrid, + "peptide nucleic acid" => Self::PeptideNucleicAcid, + "cyclic-pseudo-peptide" => Self::CyclicPseudoPeptide, + _ => Self::Other(value.to_owned()), + } + } +} + +/// Polymer description of an entity. +#[derive(Debug, Clone, Default, PartialEq, Eq, Serialize, Deserialize)] +pub struct EntityPolymer { + /// `_entity_poly.type`; `None` when the file has no `_entity_poly` row + pub kind: Option, + /// `_entity_poly_seq` by `num - 1`, unresolved residues included + pub sequence: Vec, + /// Microheterogeneous monomers after the first, keyed by `num` + pub alternatives: BTreeMap>, +} + +impl EntityPolymer { + /// Monomer at a `label_seq_id` / `_entity_poly_seq.num` position. + pub fn monomer(&self, seq_id: u32) -> Option<&str> { + let index = usize::try_from(seq_id).ok()?.checked_sub(1)?; + self.sequence.get(index).map(String::as_str) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn kinds_parse_dictionary_values_and_keep_unknown_ones() { + assert_eq!( + EntityKind::from_mmcif("non-polymer"), + EntityKind::NonPolymer + ); + assert_eq!(EntityKind::from_mmcif("Water"), EntityKind::Water); + assert_eq!( + EntityKind::from_mmcif("ligand"), + EntityKind::Other("ligand".to_owned()) + ); + assert_eq!( + PolymerKind::from_mmcif("polypeptide(L)"), + PolymerKind::PeptideL + ); + assert_eq!( + PolymerKind::from_mmcif("other"), + PolymerKind::Other("other".to_owned()) + ); + } + + #[test] + fn monomer_is_looked_up_by_one_based_position() { + let polymer = EntityPolymer { + sequence: vec!["MET".to_owned(), "GLY".to_owned()], + ..Default::default() + }; + + assert_eq!(polymer.monomer(1), Some("MET")); + assert_eq!(polymer.monomer(2), Some("GLY")); + assert_eq!(polymer.monomer(0), None); + assert_eq!(polymer.monomer(3), None); + } +} diff --git a/crates/patinae-mol/src/lib.rs b/crates/patinae-mol/src/lib.rs index 506377c..fad378d 100644 --- a/crates/patinae-mol/src/lib.rs +++ b/crates/patinae-mol/src/lib.rs @@ -48,6 +48,7 @@ mod coordset; mod dirty; pub mod dss; mod element; +mod entity; mod error; mod flags; mod index; @@ -73,6 +74,7 @@ pub use coordset::{ }; pub use dirty::DirtyFlags; pub use element::{Element, DEFAULT_COV_RADIUS, DEFAULT_VDW_RADIUS, ELEMENT_COUNT}; +pub use entity::{Entity, EntityKind, EntityPolymer, PolymerKind}; pub use error::{MolError, MolResult}; pub use flags::{AtomFlags, AtomGeometry, Chirality, Stereo}; pub use index::{AtomIndex, AtomRemap, BondIndex, CoordIndex, StateIndex, INVALID_INDEX}; diff --git a/crates/patinae-mol/src/molecule.rs b/crates/patinae-mol/src/molecule.rs index f8ba709..0e502c1 100644 --- a/crates/patinae-mol/src/molecule.rs +++ b/crates/patinae-mol/src/molecule.rs @@ -97,6 +97,10 @@ pub struct ObjectMolecule { /// File-defined assembly recipes and original chain membership. #[serde(default)] pub assembly: crate::AssemblyMetadata, + /// mmCIF entities (`_entity`, `_entity_poly`, `_entity_poly_seq`) as loaded from the + /// file; edits do not update them. Empty for other formats. + #[serde(default)] + pub entities: Vec, } impl Default for ObjectMolecule { @@ -115,6 +119,7 @@ impl Default for ObjectMolecule { symmetry: None, subchain_partition: OnceLock::new(), assembly: crate::AssemblyMetadata::default(), + entities: Vec::new(), } } } @@ -179,6 +184,7 @@ impl ObjectMolecule { symmetry: None, subchain_partition: OnceLock::new(), assembly: crate::AssemblyMetadata::default(), + entities: Vec::new(), } } diff --git a/crates/patinae-session/src/prs.rs b/crates/patinae-session/src/prs.rs index 0b0f210..5c74670 100644 --- a/crates/patinae-session/src/prs.rs +++ b/crates/patinae-session/src/prs.rs @@ -12,7 +12,7 @@ use rmpv::Value; use serde::{Deserialize, Serialize}; /// Current native PRS document format version. -pub const PRS_FORMAT_VERSION: u32 = 4; +pub const PRS_FORMAT_VERSION: u32 = 5; /// Format version assigned to legacy raw [`Session`] files. pub const PRS_LEGACY_FORMAT_VERSION: u32 = 1; @@ -361,7 +361,7 @@ fn session_value_mut(root: &mut Value) -> Result<&mut Value, PrsError> { fn validate_registry(registry: &Value, format_version: Option) -> Result<(), PrsError> { let field_count = struct_field_count(registry, "ObjectRegistrySnapshot")?; let valid_arity = match format_version { - Some(3 | 4) => field_count == 11, + Some(3..=5) => field_count == 11, Some(2) => matches!(field_count, 8 | 9), Some(1) => (7..=9).contains(&field_count), Some(version) => return Err(invalid(format!("unsupported PRS format version {version}"))), @@ -453,7 +453,7 @@ fn validate_molecules(molecules: &Value) -> Result<(), PrsError> { MOLECULE_DATA_INDEX, "MoleculeObjectSnapshot", )?; - validate_struct_arity(molecule, &[10, 11], "ObjectMolecule")?; + validate_struct_arity(molecule, &[10, 11, 12], "ObjectMolecule")?; validate_named_fields( molecule, &[ @@ -468,6 +468,7 @@ fn validate_molecules(molecules: &Value) -> Result<(), PrsError> { "unique_settings", "symmetry", "assembly", + "entities", ], &[ "atoms", @@ -953,6 +954,55 @@ mod tests { ); } + #[test] + fn version_four_without_entities_loads_and_entities_round_trip() { + let mut session = cartoon_session_with_restore(); + let entity = patinae_mol::Entity { + id: "1".to_owned(), + kind: Some(patinae_mol::EntityKind::Polymer), + polymer: Some(patinae_mol::EntityPolymer { + kind: Some(patinae_mol::PolymerKind::PeptideL), + sequence: vec!["MET".to_owned(), "GLY".to_owned()], + alternatives: [(2, vec!["ALA".to_owned()])].into(), + }), + }; + session + .registry + .get_molecule_mut("mol") + .unwrap() + .molecule_mut() + .entities = vec![entity.clone()]; + + let document = named_test_document(&session); + let restored = decode_prs_document(encode_test_value(&document)).unwrap(); + let molecule = restored.session.registry.get_molecule("mol").unwrap(); + assert_eq!(molecule.molecule().entities, [entity]); + + fn strip(value: &mut Value) { + match value { + Value::Map(fields) => { + fields.retain(|(k, _)| k.as_str() != Some("entities")); + for (_, child) in fields { + strip(child); + } + } + Value::Array(values) => { + for child in values { + strip(child); + } + } + _ => {} + } + } + let mut document = named_test_document(&session); + *test_map_field_mut(&mut document, "prs_format_version") = Value::from(4); + strip(&mut document); + let restored = decode_prs_document(encode_test_value(&document)).unwrap(); + assert_eq!(restored.prs_format_version, 4); + let molecule = restored.session.registry.get_molecule("mol").unwrap(); + assert!(molecule.molecule().entities.is_empty()); + } + #[test] fn positional_version_three_without_appended_fields_loads_as_explicit() { let mut session = cartoon_session_with_restore(); @@ -971,7 +1021,9 @@ mod tests { let owner = registry[REGISTRY_MOLECULES_INDEX].as_array_mut_for_test()[0].as_array_mut_for_test(); let snapshot = owner[1].as_array_mut_for_test(); - assert_eq!(snapshot[0].as_array_mut_for_test().len(), 11); + // Version 3 predates both trailing ObjectMolecule fields: assembly and entities. + assert_eq!(snapshot[0].as_array_mut_for_test().len(), 12); + snapshot[0].as_array_mut_for_test().pop(); snapshot[0].as_array_mut_for_test().pop(); snapshot[1].as_array_mut_for_test().pop(); for owner in registry[4].as_array_mut_for_test() { From f253c36d5935a9cee6efbd439da018a20df8942b Mon Sep 17 00:00:00 2001 From: Pavel Yakovlev Date: Thu, 8 Oct 2026 14:18:32 +0300 Subject: [PATCH 5/5] fix(ci): install Linux dependencies without the broken apt cache The APT cache hit restored only metadata and zero package archives, so Fontconfig was missing when core tests compiled Slint. Install the same Linux dependency list with apt-get and check Fontconfig before Cargo runs. --- .github/workflows/rust.yml | 14 +++++++------- 1 file changed, 7 insertions(+), 7 deletions(-) diff --git a/.github/workflows/rust.yml b/.github/workflows/rust.yml index 0a87e30..6c7d457 100644 --- a/.github/workflows/rust.yml +++ b/.github/workflows/rust.yml @@ -26,14 +26,14 @@ jobs: with: components: rustfmt, clippy - - name: Cache APT packages - uses: awalsh128/cache-apt-pkgs-action@latest - with: - packages: >- - libxkbcommon-dev libwayland-dev libxrandr-dev libxi-dev - libx11-dev libx11-xcb-dev libxcb-randr0-dev libxcb-xfixes0-dev + - name: Install Linux dependencies + run: | + sudo apt-get update + sudo apt-get install -y --no-install-recommends \ + libxkbcommon-dev libwayland-dev libxrandr-dev libxi-dev \ + libx11-dev libx11-xcb-dev libxcb-randr0-dev libxcb-xfixes0-dev \ libasound2-dev libfontconfig1-dev libgtk-3-dev libxdo-dev - version: 1.0 + pkg-config --exists fontconfig - name: Cache Rust dependencies uses: Swatinem/rust-cache@v2