From 7b4884d9e21c40607f0e08724a6b85445acdb5be Mon Sep 17 00:00:00 2001 From: Xiaoxuan Yu Date: Mon, 28 Sep 2026 14:14:57 +0800 Subject: [PATCH] feat(analysis): bind CIF source identities to H5MD atom order --- docs/source_order_binding.md | 94 +++++++++++ src/Xponge/io_bundle/source_order.py | 2 + src/XpongeCPP/analysis/cif_mdanalysis.py | 22 ++- src/XpongeCPP/io_bundle/saver.py | 46 +++--- src/XpongeCPP/io_bundle/source_order.py | 132 +++++++++++++++ tests/test_source_order_binding.py | 194 +++++++++++++++++++++++ 6 files changed, 469 insertions(+), 21 deletions(-) create mode 100644 docs/source_order_binding.md create mode 100644 src/Xponge/io_bundle/source_order.py create mode 100644 src/XpongeCPP/io_bundle/source_order.py create mode 100644 tests/test_source_order_binding.py diff --git a/docs/source_order_binding.md b/docs/source_order_binding.md new file mode 100644 index 0000000..bcec1e3 --- /dev/null +++ b/docs/source_order_binding.md @@ -0,0 +1,94 @@ +# CIF / H5MD source-order binding, version 1 + +The bundle saver binds source atom identities in its **final serialization +order** when `source_atom_ids` is supplied. XPONGE resolves IDs by Atom identity +after its preparation/reordering; XpongeCPP uses the native serialized +`molecule.atoms` order. Exports without source IDs retain the previous hashes. + +## Hash contract + +`digest(value)` is `"sha256:" + SHA256(canonical_json(value)).hexdigest()`. +Canonical JSON uses Python `json.dumps(sort_keys=True, separators=(",", ":"), +ensure_ascii=True, allow_nan=False)`, encoded as UTF-8. IDs are unique nonempty +strings. No Unicode normalization or whitespace trimming is performed. + +```text +source_atom_order_hash = digest({ + "schema": "sponge-source-atom-order-v1", + "source_atom_ids": [IDs in final simulation order] +}) + +atom_order_hash = digest({ + "schema": "sponge-bound-atom-order-v1", + "base_atom_order_hash": original native atom_order_hash, + "source_atom_order_hash": source_atom_order_hash +}) +``` + +The original native order digest is retained as `base_atom_order_hash`. Thus +permuting distinct IDs changes the bound order digest even if the swapped atoms +have identical mass, charge and residue properties. The existing topology hash +continues to identify the native physical topology payload; the added binding +and source IDs are provenance metadata. + +Before staged bundle publication, the saver writes: + +- `/topology/source_atom_ids`: UTF-8 array in final simulation order. +- `/topology/source_order_binding`: UTF-8 JSON object defined below. +- `/topology/atom_order_hash`: new bound order digest. +- Restart `/run/atom_order_hash`: the same digest. + +SPONGE already propagates this opaque topology order hash to +`/parameters/sponge/topology_compatibility/atom_order_hash` in output H5MD, +and restores global atom order before coordinate output. No per-frame IDs or +SPONGE format change is required. + +## Mapping sidecar + +Mokda checks that the serialized source IDs exactly equal the final mapping's +`external_id` sequence and that the binding agrees with current topology +metadata. It then adds an optional `topology_binding` object to the existing +`sponge-atom-order-mapping` version 1 JSON and save manifest: + +```json +{ + "schema": "sponge-source-order-binding", + "schema_version": 1, + "hash_algorithm": "sha256", + "atom_count": 2, + "base_atom_order_hash": "native exporter digest", + "source_atom_order_hash": "sha256:...", + "atom_order_hash": "sha256:...", + "topology_hash": "native topology digest" +} +``` + +This is bound during native topology export, **never by copying values from a +chosen trajectory**. Chemcore's ordered CIF and existing `mapping_hash` remain +unchanged: they prove the mapping's complete identity rows correspond to the +CIF atom order. The source-ID digest connects those rows to the native topology. + +## Analysis validation + +`load_cif_h5md_universe` checks: + +1. CIF mapping order, IDs and digest, then equality with the sidecar. +2. Binding schema and recomputed source-ID / bound-order digests. +3. CIF/H5MD atom counts. +4. Both available H5MD `atom_order_hash` and `topology_hash` against the binding. + +Conflicting available hashes or malformed bindings are rejected even with +`strict=False`. Missing sidecars, old sidecars without a binding, and older +H5MD files missing hashes remain readable with a warning and unverified status. +A custom particle stream cannot be proven by SPONGE's global hashes and is +reported as unverified. Raw exports currently have no source-order binding. + +Results expose `universe.cif_metadata["trajectory_order_validation"]` with +`verified`, `method` (when verified), or `reason` (when unverified). Mokda copies +this into result `meta.trajectoryOrderValidation` and +`analysis_inputs.json.trajectory_order_validation`. This remains available when +the original topology H5 has been removed; the CIF and sidecar suffice. + +This verifies export provenance and order consistency under the writer's +coordinate-order contract. It is not a signature, nor does it detect coordinates +edited after simulation while their provenance metadata is deliberately retained. diff --git a/src/Xponge/io_bundle/source_order.py b/src/Xponge/io_bundle/source_order.py new file mode 100644 index 0000000..5413253 --- /dev/null +++ b/src/Xponge/io_bundle/source_order.py @@ -0,0 +1,2 @@ +"""Compatibility import for source identity order bindings.""" +from XpongeCPP.io_bundle.source_order import * # noqa: F401,F403 diff --git a/src/XpongeCPP/analysis/cif_mdanalysis.py b/src/XpongeCPP/analysis/cif_mdanalysis.py index 6b91f60..67bab7e 100644 --- a/src/XpongeCPP/analysis/cif_mdanalysis.py +++ b/src/XpongeCPP/analysis/cif_mdanalysis.py @@ -15,6 +15,7 @@ from MDAnalysis.lib.util import openany from MDAnalysis.topology.base import TopologyReaderBase +from ..io_bundle.source_order import validate_source_order_binding, validate_trajectory_source_order from ..io_bundle.errors import BundleTopologyError, BundleTrajectoryError @@ -823,8 +824,17 @@ def _verify_mokda_mapping_file(mapping_file, cif_mapping): raise BundleTopologyError( "Mokda mapping file atoms do not match the CIF trajectory mapping" ) + binding = None + if "topology_binding" in document: + try: + binding = validate_source_order_binding( + document["topology_binding"], [row["external_id"] for row in sidecar_mapping] + ) + except ValueError as exc: + raise BundleTopologyError(f"invalid mapping topology_binding: {exc}") from exc return { "path": os.fspath(mapping_file), + "topology_binding": binding, "verified": True, "mapping_hash": cif_mapping_hash, "atom_count": len(cif_identity_mapping), @@ -887,6 +897,15 @@ def load_cif_h5md_universe( f"{trajectory!r} has {trajectory_atom_count} atoms, CIF topology has " f"{cif_topology.n_atoms}" ) + try: + order_validation = validate_trajectory_source_order( + trajectory, + cif_topology.cif_metadata["mapping_file_validation"].get("topology_binding"), + particle_stream=particle_stream, + ) + except ValueError as exc: + raise BundleTrajectoryError(str(exc)) from exc + cif_topology.cif_metadata["trajectory_order_validation"] = order_validation if cif_topology.cif_metadata["mokda_trajectory_mapping"]: mapping_file_verified = cif_topology.cif_metadata["mapping_file_validation"][ "verified" @@ -906,7 +925,8 @@ def load_cif_h5md_universe( "CIF/H5MD compatibility is checked by atom count only; ensure the CIF " "atom order matches the trajectory atom order." ) - warnings.warn(warning, RuntimeWarning, stacklevel=2) + if not order_validation["verified"]: + warnings.warn(warning, RuntimeWarning, stacklevel=2) universe = mda.Universe( cif_topology, diff --git a/src/XpongeCPP/io_bundle/saver.py b/src/XpongeCPP/io_bundle/saver.py index a90e045..efd91b3 100644 --- a/src/XpongeCPP/io_bundle/saver.py +++ b/src/XpongeCPP/io_bundle/saver.py @@ -16,6 +16,7 @@ ) from .errors import BundlePathError from .protocol import SpongeProtocol, add_protocol_to_bundle +from .source_order import bind_bundle_source_order def save_sponge_input_bundle( @@ -65,6 +66,29 @@ def save_sponge_input_bundle( "ResidueType, or registered template-like object" ) + values = None + if return_mapping and source_atom_ids is None: + raise ValueError("return_mapping=True requires source_atom_ids") + if source_atom_ids is not None: + atoms = list(target.atoms) + if isinstance(source_atom_ids, dict): + by_index = { + int(atom.index): str(value) for atom, value in source_atom_ids.items() + } + if set(by_index) != {int(atom.index) for atom in atoms}: + raise ValueError( + "source_atom_ids mapping must cover every input Atom exactly once" + ) + values = tuple(by_index[int(atom.index)] for atom in atoms) + else: + values = tuple(str(value) for value in source_atom_ids) + if len(values) != len(atoms): + raise ValueError( + "source_atom_ids must contain one ID for every input Atom" + ) + if len(set(values)) != len(values): + raise ValueError("source_atom_ids must be unique") + output_root = Path(dirname).resolve() output_root.mkdir(parents=True, exist_ok=True) normalized_prefix = str(prefix or getattr(molecule, "name", "system")) @@ -79,6 +103,8 @@ def save_sponge_input_bundle( target, staged_prefix, str(staging_root), protocol=None ) staged_paths = _prefixed_bundle_paths(staging_root, staged_prefix) + if values is not None: + bind_bundle_source_order(staged_paths, values) _apply_protocol(staged_paths, protocol) for source, destination in ( (staged_paths.topology, final_paths.topology), @@ -90,26 +116,6 @@ def save_sponge_input_bundle( del prepared if not return_mapping: return target - if source_atom_ids is None: - raise ValueError("return_mapping=True requires source_atom_ids") - atoms = list(target.atoms) - if isinstance(source_atom_ids, dict): - by_index = { - int(atom.index): str(value) for atom, value in source_atom_ids.items() - } - if set(by_index) != {int(atom.index) for atom in atoms}: - raise ValueError( - "source_atom_ids mapping must cover every input Atom exactly once" - ) - values = tuple(by_index[int(atom.index)] for atom in atoms) - else: - values = tuple(str(value) for value in source_atom_ids) - if len(values) != len(atoms): - raise ValueError( - "source_atom_ids must contain one ID for every input Atom" - ) - if len(set(values)) != len(values): - raise ValueError("source_atom_ids must be unique") return target, tuple( {"simulation_index": index, "source_atom_id": source_id} for index, source_id in enumerate(values) diff --git a/src/XpongeCPP/io_bundle/source_order.py b/src/XpongeCPP/io_bundle/source_order.py new file mode 100644 index 0000000..7da3026 --- /dev/null +++ b/src/XpongeCPP/io_bundle/source_order.py @@ -0,0 +1,132 @@ +"""Versioned binding of serialized source identities to native atom order.""" + +from __future__ import annotations + +import hashlib +import json + +import h5py +import numpy as np + + +def _digest(value): + payload = json.dumps(value, sort_keys=True, separators=(",", ":"), allow_nan=False) + return "sha256:" + hashlib.sha256(payload.encode("utf-8")).hexdigest() + + +def make_source_order_binding(source_atom_ids, base_atom_order_hash, topology_hash): + """Bind the actual serialized identity sequence to its native order hash.""" + identities = list(source_atom_ids) + if not identities or any(not isinstance(value, str) or not value for value in identities): + raise ValueError("source atom IDs must be nonempty strings") + if len(set(identities)) != len(identities): + raise ValueError("source atom IDs must be unique") + if not all(isinstance(value, str) and value for value in (base_atom_order_hash, topology_hash)): + raise ValueError("source order binding requires native topology and atom-order hashes") + source_hash = _digest({"schema": "sponge-source-atom-order-v1", "source_atom_ids": identities}) + order_hash = _digest({ + "schema": "sponge-bound-atom-order-v1", + "base_atom_order_hash": base_atom_order_hash, + "source_atom_order_hash": source_hash, + }) + return { + "schema": "sponge-source-order-binding", + "schema_version": 1, + "hash_algorithm": "sha256", + "atom_count": len(identities), + "base_atom_order_hash": base_atom_order_hash, + "source_atom_order_hash": source_hash, + "atom_order_hash": order_hash, + "topology_hash": topology_hash, + } + + +def validate_source_order_binding(binding, source_atom_ids): + """Reject malformed bindings and identity sequences different from export.""" + if not isinstance(binding, dict): + raise ValueError("source order binding must be an object") + expected = make_source_order_binding( + source_atom_ids, binding.get("base_atom_order_hash"), binding.get("topology_hash") + ) + if binding != expected: + raise ValueError("source order binding does not match the atom identity sequence or schema") + return expected + + +def _text(value): + return value.decode("utf-8") if isinstance(value, bytes) else str(value) + + +def _write_text(handle, path, value): + if path in handle: + del handle[path] + handle.create_dataset(path, data=value, dtype=h5py.string_dtype("utf-8")) + + +def bind_bundle_source_order(paths, source_atom_ids): + """Bind a staged bundle before publication; keep restart lineage consistent. + + Only the exporter may call this with IDs in its actual serialization order. + No coordinates or force-field datasets are changed. + """ + identities = list(source_atom_ids) + with h5py.File(paths.topology, "r+") as top, h5py.File(paths.restart, "r+") as restart: + if "/topology/source_order_binding" in top: + raise ValueError("bundle source atom order is already bound") + count = int(np.asarray(top["/topology/atom_count"][()]).reshape(-1)[0]) + if len(identities) != count: + raise ValueError("source atom IDs must cover every serialized atom") + binding = make_source_order_binding( + identities, + _text(top["/topology/atom_order_hash"][()]), + _text(top["/topology/topology_hash"][()]), + ) + top.create_dataset("/topology/source_atom_ids", data=identities, dtype=h5py.string_dtype("utf-8")) + _write_text(top, "/topology/source_order_binding", json.dumps(binding, sort_keys=True)) + _write_text(top, "/topology/atom_order_hash", binding["atom_order_hash"]) + _write_text(restart, "/run/atom_order_hash", binding["atom_order_hash"]) + return binding + + +def read_source_order_binding(topology, source_atom_ids): + """Read an export binding, checking its stored IDs and current H5 metadata.""" + with h5py.File(topology, "r") as top: + if "/topology/source_order_binding" not in top: + return None + binding = validate_source_order_binding( + json.loads(_text(top["/topology/source_order_binding"][()])), source_atom_ids + ) + if top["/topology/source_atom_ids"].asstr()[...].tolist() != list(source_atom_ids): + raise ValueError("topology source atom IDs do not match the final mapping") + for name in ("atom_order_hash", "topology_hash"): + if _text(top[f"/topology/{name}"][()]) != binding[name]: + raise ValueError(f"topology {name} does not match its source order binding") + count = int(np.asarray(top["/topology/atom_count"][()]).reshape(-1)[0]) + if count != binding["atom_count"]: + raise ValueError("topology atom count does not match its source order binding") + return binding + + +def validate_trajectory_source_order(trajectory, binding, *, particle_stream=None): + """Check available trajectory hashes; legacy missing hashes stay unverified. + + SPONGE compatibility hashes describe the global ``all`` particle stream. + They cannot establish the identity order of an arbitrary custom stream. + """ + if binding is None: + return {"verified": False, "reason": "missing_source_order_binding"} + missing = [] + with h5py.File(trajectory, "r") as handle: + streams = list(handle.get("particles", {})) + selected = particle_stream or (streams[0] if len(streams) == 1 else None) + if selected != "all": + return {"verified": False, "reason": "unbound_particle_stream"} + for name in ("atom_order_hash", "topology_hash"): + path = f"/parameters/sponge/topology_compatibility/{name}" + if path not in handle or not _text(handle[path][()]): + missing.append(name) + elif _text(handle[path][()]) != binding[name]: + raise ValueError(f"H5MD {name} does not match the CIF mapping source order binding") + if missing: + return {"verified": False, "reason": "missing_trajectory_hash", "missing": missing} + return {"verified": True, "method": "source_order_binding_v1", **binding} diff --git a/tests/test_source_order_binding.py b/tests/test_source_order_binding.py new file mode 100644 index 0000000..d6a6e46 --- /dev/null +++ b/tests/test_source_order_binding.py @@ -0,0 +1,194 @@ +import importlib +import json +import warnings +from types import SimpleNamespace + +import h5py +import numpy as np +import pytest + +import Xponge +from Xponge.analysis import md_analysis as xmda +from Xponge.io_bundle.errors import BundleTopologyError, BundleTrajectoryError +from Xponge.io_bundle.source_order import ( + bind_bundle_source_order, make_source_order_binding, read_source_order_binding, + validate_source_order_binding, validate_trajectory_source_order, +) + + +def _text(handle, path, value): + if path in handle: + del handle[path] + handle.create_dataset(path, data=value, dtype=h5py.string_dtype('utf-8')) + + +def _bundle(tmp_path): + paths = SimpleNamespace(topology=tmp_path/'top.h5', restart=tmp_path/'rst.h5') + with h5py.File(paths.topology, 'w') as f: + f['/topology/atom_count'] = 2 + _text(f, '/topology/atom_order_hash', 'native-order') + _text(f, '/topology/topology_hash', 'native-topology') + with h5py.File(paths.restart, 'w') as f: + _text(f, '/run/atom_order_hash', 'native-order') + return paths + + +def _trajectory(path, binding, stream='all'): + with h5py.File(path, 'w') as f: + f.require_group('/h5md').attrs['version'] = [1, 1] + f.require_group('/h5md/creator').attrs.update(name='SPONGE', version='test') + _text(f, '/parameters/sponge/schema/name', 'sponge.output.h5md') + _text(f, '/parameters/sponge/schema/version', 'sponge.output.v2') + _text(f, '/parameters/sponge/output/status', 'finalized') + for key in ('atom_order_hash', 'topology_hash'): + _text(f, '/parameters/sponge/topology_compatibility/'+key, binding[key]) + p = f.require_group('/particles/'+stream) + p['step'] = [0, 1] + p['time'] = [0., 1.] + p['time'].attrs['unit'] = 'ps' + position = p.create_group('position') + position['value'] = np.zeros((2, 2, 3), dtype=np.float32) + position['value'].attrs['unit'] = 'Angstrom' + position['step'], position['time'] = p['step'], p['time'] + box = p.create_group('box') + box.attrs.update(dimension=3, boundary=np.asarray(['periodic']*3, dtype='S8')) + edges = box.create_group('edges') + edges['value'] = np.asarray([np.eye(3)*10]*2) + edges['value'].attrs['unit'] = 'Angstrom' + edges['step'], edges['time'] = p['step'], p['time'] + + +def _cif_pair(tmp_path): + import hashlib + atoms = [dict(simulation_index=i, external_id=f'atom:{i}', canonical_atom_id=i+1, + simulation_residue_index=0, simulation_residue_id='res:0') for i in range(2)] + digest = hashlib.sha256(json.dumps({'atom_mapping': atoms}, sort_keys=True, + separators=(',', ':')).encode()).hexdigest() + cif = tmp_path/'system_trajectory_topology.cif' + tags = ['id', 'type_symbol', 'label_atom_id', 'label_comp_id', 'label_asym_id', + 'label_seq_id', 'Cartn_x', 'Cartn_y', 'Cartn_z'] + mapping_tags = ['simulation_index', 'atom_site_id', 'canonical_atom_id', 'external_id', + 'simulation_residue_index', 'simulation_residue_id', 'mapping_hash'] + cif.write_text('data_test\nloop_\n' + '\n'.join('_atom_site.'+tag for tag in tags) + + '\n1 C C1 MOL A 1 0 0 0\n2 C C2 MOL A 1 1 0 0\n#\nloop_\n' + + '\n'.join('_mokda_trajectory_mapping.'+tag for tag in mapping_tags) + + '\n' + '\n'.join(f'{i} {i+1} {i+1} atom:{i} 0 res:0 {digest}' for i in range(2))+'\n') + binding = make_source_order_binding(['atom:0', 'atom:1'], 'native-order', 'native-topology') + document = dict(schema='sponge-atom-order-mapping', schema_version=1, + mapping_hash=digest, atoms=atoms, topology_binding=binding) + sidecar = tmp_path/'system_atom_order_mapping.json' + sidecar.write_text(json.dumps(document)) + trajectory = tmp_path/'trajectory.h5md' + _trajectory(trajectory, binding) + return cif, sidecar, trajectory, document + + +def test_binding_changes_for_identity_permutation(): + a = make_source_order_binding(['C1', 'C2'], 'same-physical-properties', 'same-topology') + b = make_source_order_binding(['C2', 'C1'], 'same-physical-properties', 'same-topology') + assert a['atom_order_hash'] != b['atom_order_hash'] + assert validate_source_order_binding(a, ['C1', 'C2']) == a + with pytest.raises(ValueError, match='identity sequence'): + validate_source_order_binding(a, ['C2', 'C1']) + + +@pytest.mark.parametrize('ids', [[], [''], ['a', 'a'], ['a', None]]) +def test_binding_rejects_invalid_identities(ids): + with pytest.raises(ValueError): + make_source_order_binding(ids, 'order', 'topology') + + +def test_staged_binding_updates_restart_and_checks_metadata(tmp_path): + paths = _bundle(tmp_path) + binding = bind_bundle_source_order(paths, ['a', 'b']) + assert read_source_order_binding(paths.topology, ['a', 'b']) == binding + with h5py.File(paths.restart) as f: + assert f['/run/atom_order_hash'].asstr()[()] == binding['atom_order_hash'] + with pytest.raises(ValueError): + read_source_order_binding(paths.topology, ['b', 'a']) + with h5py.File(paths.topology, 'r+') as f: + _text(f, '/topology/atom_order_hash', 'wrong') + with pytest.raises(ValueError, match='atom_order_hash'): + read_source_order_binding(paths.topology, ['a', 'b']) + + +def test_cif_h5md_bound_without_native_topology(tmp_path): + cif, _, trajectory, _ = _cif_pair(tmp_path) + with warnings.catch_warnings(record=True) as caught: + warnings.simplefilter('always') + universe = xmda.load_cif_h5md_universe(cif, trajectory) + try: + assert universe.cif_metadata['trajectory_order_validation']['verified'] + assert not any('CIF/H5MD compatibility' in str(w.message) for w in caught) + finally: + universe.trajectory.close() + + +@pytest.mark.parametrize('field', ['atom_order_hash', 'topology_hash']) +def test_cif_rejects_same_count_wrong_trajectory(tmp_path, field): + cif, _, trajectory, document = _cif_pair(tmp_path) + with h5py.File(trajectory, 'r+') as f: + _text(f, '/parameters/sponge/topology_compatibility/'+field, 'other-system') + with pytest.raises(BundleTrajectoryError, match=field): + xmda.load_cif_h5md_universe(cif, trajectory, strict=False) + + +@pytest.mark.parametrize('change', ['permuted', 'version', 'algorithm', 'null', 'count']) +def test_cif_rejects_invalid_sidecar_binding(tmp_path, change): + cif, sidecar, trajectory, document = _cif_pair(tmp_path) + if change == 'permuted': + document['topology_binding'] = make_source_order_binding(['atom:1','atom:0'], 'native-order', 'native-topology') + elif change == 'null': + document['topology_binding'] = None + else: + key = {'version': 'schema_version', 'algorithm': 'hash_algorithm', 'count': 'atom_count'}[change] + document['topology_binding'][key] = 'invalid' + sidecar.write_text(json.dumps(document)) + with pytest.raises(BundleTopologyError, match='topology_binding'): + xmda.load_cif_h5md_universe(cif, trajectory) + + +@pytest.mark.parametrize('legacy', ['no_sidecar', 'old_sidecar', 'missing_hash']) +def test_legacy_inputs_still_load_unverified(tmp_path, legacy): + cif, sidecar, trajectory, document = _cif_pair(tmp_path) + if legacy == 'no_sidecar': + sidecar.unlink() + elif legacy == 'old_sidecar': + del document['topology_binding'] + sidecar.write_text(json.dumps(document)) + else: + with h5py.File(trajectory, 'r+') as f: + del f['/parameters/sponge/topology_compatibility/atom_order_hash'] + with pytest.warns(RuntimeWarning, match='CIF/H5MD compatibility'): + universe = xmda.load_cif_h5md_universe(cif, trajectory) + assert not universe.cif_metadata['trajectory_order_validation']['verified'] + universe.trajectory.close() + + +def test_global_hash_does_not_verify_custom_particle_stream(tmp_path): + binding = make_source_order_binding(['a', 'b'], 'order', 'topology') + trajectory = tmp_path/'custom.h5md' + _trajectory(trajectory, binding, stream='subset') + assert not validate_trajectory_source_order(trajectory, binding)['verified'] + + +def test_real_saver_binds_actual_serialization_order(tmp_path): + importlib.import_module('Xponge.forcefield.amber.ff14sb') + molecule = Xponge.get_peptide_from_sequence('AA') + if hasattr(molecule, 'get_atoms'): + molecule.get_atoms() + ids = [f'atom:{i}' for i, _ in enumerate(molecule.atoms)] + bindings = [] + for prefix, identities in [('first', ids), ('second', list(reversed(ids)))]: + _, mapping = Xponge.save_sponge_input(molecule, prefix, dirname=str(tmp_path), + format='bundle', source_atom_ids=identities, + return_mapping=True) + serialized = [row['source_atom_id'] for row in mapping] + topology = tmp_path/f'{prefix}_topology.spgt.h5' + binding = read_source_order_binding(topology, serialized) + assert binding is not None + with h5py.File(tmp_path/f'{prefix}_restart.spgr.h5') as f: + assert f['/run/atom_order_hash'].asstr()[()] == binding['atom_order_hash'] + bindings.append(binding) + assert bindings[0]['base_atom_order_hash'] == bindings[1]['base_atom_order_hash'] + assert bindings[0]['atom_order_hash'] != bindings[1]['atom_order_hash']