Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
34 commits
Select commit Hold shift + click to select a range
b465834
feat(dpmodel): add Uni-Mol v1 encoder blocks and an exact GELU
iProzd Sep 12, 2026
a6a196f
feat(dpmodel): add the Uni-Mol v1 data-side transforms
iProzd Sep 12, 2026
4591035
feat: register the exact GELU in every backend activation table
iProzd Sep 12, 2026
0fecbd6
feat(dpmodel): add the Uni-Mol v1 pretraining heads and loss
iProzd Sep 12, 2026
c6b85ef
feat(dpmodel): add the unimol descriptor
iProzd Sep 12, 2026
df72506
feat: import released Uni-Mol v1 checkpoints
iProzd Sep 12, 2026
bc53505
feat(dpmodel): add the unimol_pretrain fitting
iProzd Sep 12, 2026
d488e55
feat: register the unimol descriptor, fitting and loss in argcheck
iProzd Sep 12, 2026
dbd8ab3
feat(dpmodel): register the unimol_pretrain model
iProzd Sep 12, 2026
24f46d4
feat(pt-expt): wire the unimol model into the PyTorch-Exportable backend
iProzd Sep 12, 2026
e6fb4d7
test: parity tests for the Uni-Mol v1 port
iProzd Sep 12, 2026
b77b533
feat(dpmodel): apply Uni-Mol's dropout during training
iProzd Sep 12, 2026
d98f85a
feat: convert Uni-Mol pretraining data into a deepmd dataset
iProzd Sep 12, 2026
8042bec
feat: make Uni-Mol pretraining reachable from a training run
iProzd Sep 12, 2026
e69c6af
fix(dpmodel): place every constructed array on the input's device
iProzd Sep 12, 2026
04d917d
fix: refuse a periodic cell at the model boundary
iProzd Sep 12, 2026
63251e2
feat: let an objective declare the data transform it needs
iProzd Sep 12, 2026
18649c0
docs: describe how a Uni-Mol run is configured, and validate the example
iProzd Sep 12, 2026
5d4043d
fix: make `dp --pt-expt train` work for the Uni-Mol objective
iProzd Sep 12, 2026
9aa9930
test: train from a configuration, the way the feature is used
iProzd Sep 12, 2026
f333398
fix: say what the locality guard actually rules out
iProzd Sep 12, 2026
d0c192d
feat: expose the Adam epsilon
iProzd Sep 12, 2026
6ae1a1e
fix(dpmodel): train the token embedding and the distance basis
iProzd Sep 12, 2026
7fbcd2e
fix: corruption varies per epoch, and cropping moves to the converter
iProzd Sep 12, 2026
fe88321
fix: keep the embedding and the distance basis in the gradient
iProzd Sep 12, 2026
8cc8e69
fix: act on the review of this pull request
iProzd Sep 12, 2026
7f486c1
fix: keep the objective finite when nothing is corrupted
iProzd Sep 12, 2026
ee98665
fix: let the corruption survive a decoder worker
iProzd Sep 12, 2026
e82b575
fix: honour a precision the backbone does not share
iProzd Sep 12, 2026
aacaa7e
feat: match the published LMDB layout, and train in single precision
iProzd Sep 13, 2026
3b48dd1
fix: act on the review of this pull request
iProzd Sep 14, 2026
92ef3ef
fix: act on the second review of this pull request
iProzd Sep 16, 2026
4f43a33
test: cover the two review paths that had no test
iProzd Sep 16, 2026
a309c88
fix: give `xp_erf` a TensorFlow branch
iProzd Sep 22, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions deepmd/common.py
Original file line number Diff line number Diff line change
Expand Up @@ -53,6 +53,7 @@
"tanh",
"gelu",
"gelu_tf",
"gelu_erf",
"silu",
"silut",
"none",
Expand Down
44 changes: 44 additions & 0 deletions deepmd/dpmodel/array_api.py
Original file line number Diff line number Diff line change
Expand Up @@ -516,6 +516,50 @@ def xp_sigmoid(x: Array) -> Array:
return 1 / (1 + xp.exp(-x))


def xp_erf(x: Array) -> Array:
"""Compute the error function.

Used by the exact (non-approximated) GELU. The array API has no ``erf``, so
each backend's own implementation is used; NumPy goes through SciPy, which
is already a core dependency.
"""
if array_api_compat.is_jax_array(x):
from deepmd.jax.env import (
jax,
)

return jax.scipy.special.erf(x)
elif array_api_compat.is_torch_array(x):
import torch

return torch.special.erf(x)

xp = array_api_compat.array_namespace(x)
if getattr(xp, "__name__", "") == "deepmd._vendors.ndtensorflow":
import tensorflow as tf

# The NumPy round-trip below cannot serve TensorFlow. Under
# ``tf.function`` the conversion is refused outright, and in eager mode
# it detaches the erf factor from the tape, which leaves the exact GELU
# differentiating to Phi(x) alone -- silently, and for every backend
# user of ``gelu_erf`` rather than only Uni-Mol.
#
# Imported directly rather than through ``deepmd.tf2.env`` for symmetry
# with the JAX branch above: that module raises at import time unless
# eager execution is on, and this branch has to work inside
# ``tf.function``, where it is not.
return xp.asarray(tf.math.erf(x.unwrap()))

from scipy.special import (
erf,
)

if array_api_compat.is_numpy_array(x):
return erf(x)
# array-api-strict and friends: round-trip through NumPy.
return xp.asarray(erf(np.asarray(x)), dtype=x.dtype)
Comment thread
iProzd marked this conversation as resolved.


def xp_setitem_at(x: Array, mask: Array, values: Array) -> Array:
"""Set items at boolean mask indices.

Expand Down
4 changes: 4 additions & 0 deletions deepmd/dpmodel/atomic_model/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -45,6 +45,9 @@
from .property_atomic_model import (
DPPropertyAtomicModel,
)
from .unimol_atomic_model import (
DPUniMolAtomicModel,
)

__all__ = [
"BaseAtomicModel",
Expand All @@ -54,6 +57,7 @@
"DPEnergyAtomicModel",
"DPPolarAtomicModel",
"DPPropertyAtomicModel",
"DPUniMolAtomicModel",
"DPZBLLinearEnergyAtomicModel",
"LinearEnergyAtomicModel",
"PairTabAtomicModel",
Expand Down
106 changes: 106 additions & 0 deletions deepmd/dpmodel/atomic_model/unimol_atomic_model.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,106 @@
# SPDX-License-Identifier: LGPL-3.0-or-later
"""Atomic model for Uni-Mol v1 self-supervised pretraining."""

from typing import (
Any,
)

from deepmd.dpmodel.array_api import (
Array,
)
from deepmd.dpmodel.descriptor.unimol import (
DescrptUniMol,
)
from deepmd.dpmodel.fitting.unimol_pretrain import (
UniMolPretrainFitting,
)

from .dp_atomic_model import (
DPAtomicModel,
)


class DPUniMolAtomicModel(DPAtomicModel):
r"""Uni-Mol pretraining, wired at token resolution.

The standard path hands a descriptor's five-tuple to a fitting, which
cannot carry the two virtual tokens, the pair channel or the norm
regularisers that these heads read. This model therefore overrides one
method to route the backbone's token-resolution output straight into the
heads. Nothing else about the atomic model changes.
"""

def __init__(
self, descriptor: Any, fitting: Any, type_map: list[str], **kwargs: Any
) -> None:
if not isinstance(descriptor, DescrptUniMol):
raise TypeError(
"DPUniMolAtomicModel needs the unimol descriptor, which is the only "
"one producing a Uni-Mol token sequence"
)
if not isinstance(fitting, UniMolPretrainFitting):
raise TypeError("DPUniMolAtomicModel needs the unimol_pretrain fitting")
# The objective compares against distances whose virtual tokens sit at
# the origin, which is where upstream puts them and, once the transform
# has centred a frame, where the clean centroid is. Under "centroid" the
# descriptor instead places them at the centroid of the coordinates it
# is handed -- the corrupted ones -- so the two virtual columns of every
# corrupted row would be regressed against a label for a different
# position, by about the size of the noise. Centring costs nothing here
# because the transform always centres, so the only effect would be that
# silent mismatch.
if getattr(descriptor, "virtual_token_position", "origin") != "origin":
raise ValueError(
"unimol pretraining needs virtual_token_position='origin': the "
"distance target places the virtual tokens at the origin, and "
f"this descriptor places them at the "
f"{descriptor.virtual_token_position}, so the two virtual "
"columns would train against the wrong label. The corruption "
"centres every frame, so 'origin' is the centroid anyway"
)
super().__init__(descriptor, fitting, type_map, **kwargs)

def forward_atomic(
self,
extended_coord: Array,
extended_atype: Array,
nlist: Array,
mapping: Array | None = None,
fparam: Array | None = None,
aparam: Array | None = None,
comm_dict: dict | None = None,
charge_spin: Array | None = None,
) -> dict[str, Array]:
"""Run the backbone and its heads at token resolution.

Parameters
----------
extended_coord
nf x (nall x 3) coordinates; the descriptor rejects any frame that
carries periodic images.
extended_atype
nf x nall element types, already clamped to be non-negative.
nlist
nf x nloc x nnei neighbour list, which is how real atoms are told
apart from padding.
mapping, fparam, aparam, comm_dict, charge_spin
Unused by this model.

Returns
-------
dict
The three head outputs plus the two norm regularisers.
"""
del mapping, fparam, aparam, comm_dict, charge_spin
backbone = self.descriptor.forward_tokens(extended_coord, extended_atype, nlist)
return self.fitting_net.call_tokens(backbone)

def apply_out_stat(self, ret: dict[str, Array], atype: Array) -> dict[str, Array]:
"""Return the head outputs untouched.

Self-supervised targets carry no per-element bias to add back: the
element head predicts a distribution, and the coordinate and distance
heads predict geometry the data already fixes.
"""
del atype
return ret
4 changes: 4 additions & 0 deletions deepmd/dpmodel/descriptor/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -35,6 +35,9 @@
from .se_t_tebd import (
DescrptSeTTebd,
)
from .unimol import (
DescrptUniMol,
)

__all__ = [
"DescrptDPA1",
Expand All @@ -48,5 +51,6 @@
"DescrptSeR",
"DescrptSeT",
"DescrptSeTTebd",
"DescrptUniMol",
"make_base_descriptor",
]
Loading
Loading