Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
8 changes: 8 additions & 0 deletions .gitignore
Original file line number Diff line number Diff line change
Expand Up @@ -206,6 +206,10 @@ tempCodeRunnerFile.py
# Ruff stuff:
.ruff_cache/

# step-up training artifacts and local data
outputs/
data/processed/

# PyPI configuration file
.pypirc

Expand All @@ -216,3 +220,7 @@ __marimo__/

# Streamlit
.streamlit/secrets.toml

# Slurm job logs (scripts/train.sh writes train-<jobid>.out into the submit dir)
*-[0-9][0-9][0-9][0-9][0-9][0-9][0-9].out
slurm-*.out
6 changes: 6 additions & 0 deletions .gitmodules
Original file line number Diff line number Diff line change
@@ -0,0 +1,6 @@
[submodule "external/ReBIND"]
path = external/ReBIND
url = https://github.com/holymollyhao/ReBIND.git
[submodule "external/GTMGC"]
path = external/GTMGC
url = https://github.com/Rich-XGK/GTMGC.git
7 changes: 5 additions & 2 deletions CONTRIBUTING.md
Original file line number Diff line number Diff line change
Expand Up @@ -4,14 +4,17 @@ Thanks for helping improve `step-up`.

## Development Setup

Clone the repository and install dependencies:
Clone the repository (with `--recursive`, so the vendored ReBind submodule under
`external/ReBIND` is checked out) and install dependencies:

```bash
git clone https://github.com/LeMaterial/step-up.git
git clone --recursive https://github.com/LeMaterial/step-up.git
cd step-up
uv sync --dev
```

If you already cloned without `--recursive`, run `git submodule update --init --recursive`.

Install the pre-commit hook:

```bash
Expand Down
380 changes: 372 additions & 8 deletions README.md

Large diffs are not rendered by default.

50 changes: 50 additions & 0 deletions configs/bostmc.yaml
Original file line number Diff line number Diff line change
@@ -0,0 +1,50 @@
# Full BOSTMC low-spin training run: singlets and doublets together (121,496
# complexes), with the model conditioned on charge and spin so the two manifolds
# are distinguishable.
#
# Data is the filtered release, which drops structures whose molecular graph
# changed during optimization and deduplicates by initial graph hash, keeping the
# best-R-factor representative.
#
# Split: the project's own random split (seed 0, 80/10/10 by refcode) from
# datasets/filtered/splits/random, which covers all 121,496 low-spin rows. The
# similarity ("spectral") split under splits/spectral is the planned follow-up.
dataset_path: /home/gridsan/jtoney/BOSTMC/datasets/filtered/BOSTMC-low-spin.csv
dataset_source: mol2
subset_size: null
id_column: refcode
split_files:
train: /home/gridsan/jtoney/BOSTMC/datasets/filtered/splits/random/train-random.csv
val: /home/gridsan/jtoney/BOSTMC/datasets/filtered/splits/random/val-random.csv
test: /home/gridsan/jtoney/BOSTMC/datasets/filtered/splits/random/test-random.csv

# Charges span -8..+8 and spin multiplicity is 1 or 2. Both enter as scalars, so
# rare charge states still inform the model and inference isn't restricted to the
# combinations seen in training.
charge_column: charge
spin_column: spinmult

n_layers: 8
d_model: 512
d_ffn: 1024
n_head: 8
dropout: 0.0

# Optimization as in ReBind's rebind.sh (see README for the fp16 exception).
epochs: 20
batch_size: 100
eval_batch_size: 100
lr: 9.0e-5
weight_decay: 0.0
warmup_ratio: 0.1
adam_beta1: 0.9
adam_beta2: 0.99
adam_eps: 1.0e-8
lr_schedule: linear
grad_clip: 1.0
drop_last: true
num_workers: 4

device: cuda
output_dir: outputs/bostmc_full
seed: 42
58 changes: 58 additions & 0 deletions configs/bostmc_noh.yaml
Original file line number Diff line number Diff line change
@@ -0,0 +1,58 @@
# Heavy-atom BOSTMC: configs/bostmc.yaml with hydrogens dropped.
# complexes), with the model conditioned on charge and spin so the two manifolds
# are distinguishable.
#
# Data is the filtered release, which drops structures whose molecular graph
# changed during optimization and deduplicates by initial graph hash, keeping the
# best-R-factor representative.
#
# Split: the project's own random split (seed 0, 80/10/10 by refcode) from
# datasets/filtered/splits/random, which covers all 121,496 low-spin rows. The
# similarity ("spectral") split under splits/spectral is the planned follow-up.
dataset_path: /home/gridsan/jtoney/BOSTMC/datasets/filtered/BOSTMC-low-spin.csv
dataset_source: mol2
subset_size: null
id_column: refcode
split_files:
train: /home/gridsan/jtoney/BOSTMC/datasets/filtered/splits/random/train-random.csv
val: /home/gridsan/jtoney/BOSTMC/datasets/filtered/splits/random/val-random.csv
test: /home/gridsan/jtoney/BOSTMC/datasets/filtered/splits/random/test-random.csv

# Charges span -8..+8 and spin multiplicity is 1 or 2. Both enter as scalars, so
# rare charge states still inform the model and inference isn't restricted to the
# combinations seen in training.
charge_column: charge
spin_column: spinmult

n_layers: 8
d_model: 512
d_ffn: 1024
n_head: 8
dropout: 0.0

# Optimization as in ReBind's rebind.sh (see README for the fp16 exception).
epochs: 20
batch_size: 100
eval_batch_size: 100
lr: 9.0e-5
weight_decay: 0.0
warmup_ratio: 0.1
adam_beta1: 0.9
adam_beta2: 0.99
adam_eps: 1.0e-8
lr_schedule: linear
grad_clip: 1.0
drop_last: true
num_workers: 4

device: cuda
output_dir: outputs/bostmc_full_noh
seed: 42

# Heavy-atom variant: hydrogens are dropped from the graph, so the model never
# sees or predicts them. Paired with configs/bostmc.yaml to separate "the model is
# bad at hydrogen" from "the model is bad at geometry". Each heavy atom keeps its
# hydrogen count in the numH feature, so the graph is not missing the chemistry,
# only the positions. Note this departs from the published setup, which keeps
# hydrogen explicit, so it is not a reproduction run.
remove_hs: true
31 changes: 31 additions & 0 deletions configs/bostmc_smoke.yaml
Original file line number Diff line number Diff line change
@@ -0,0 +1,31 @@
# BOSTMC smoke run on the first 100 low-spin rows (like bostmc.yaml). Verifies the
# direct MOL2 parser end-to-end (no RDKit), the LJ patch on d-block elements, and
# the charge/spin conditioning path.
dataset_path: /home/gridsan/jtoney/BOSTMC/datasets/filtered/BOSTMC-low-spin.csv
dataset_source: mol2
subset_size: 100
# Split key: refcode.
id_column: refcode
charge_column: charge
spin_column: spinmult
cache_dataset: true
split_ratios: [0.8, 0.1, 0.1]
split_seed: 0

n_layers: 2
d_model: 64
d_ffn: 128
n_head: 4
dropout: 0.0

epochs: 3
batch_size: 4
eval_batch_size: 4
lr: 3.0e-4
weight_decay: 0.0
warmup_ratio: 0.1
num_workers: 0

device: cpu
output_dir: outputs/bostmc_smoke
seed: 0
29 changes: 29 additions & 0 deletions configs/qm9.yaml
Original file line number Diff line number Diff line change
@@ -0,0 +1,29 @@
# Full QM9-full.csv training run (~134K molecules).
# GPU-only, staged for the Slurm job once compute is available.
# Follows ReBind's QM9 setup from external/ReBIND/experiments/conformer_prediction/rebind.sh
# (see README for the few deliberate differences).
dataset_path: /home/gridsan/jtoney/ElemeNet-benchmarking/benchmarking/datasets/QM9-full.csv
dataset_source: smiles
subset_size: null
# Split key: mol_id, so QM9's few duplicated rows always share a split.
id_column: mol_id
split_ratios: [0.9, 0.05, 0.05]
split_seed: 0

n_layers: 8
d_model: 512
d_ffn: 1024
n_head: 8
dropout: 0.0

epochs: 20
batch_size: 100
eval_batch_size: 100
lr: 9.0e-5
weight_decay: 0.0
warmup_ratio: 0.1
num_workers: 4

device: cuda
output_dir: outputs/qm9_full
seed: 0
50 changes: 50 additions & 0 deletions configs/qm9_gtmgc.yaml
Original file line number Diff line number Diff line change
@@ -0,0 +1,50 @@
# GTMGC (Xu et al., ICLR 2024) on QM9, same data and split as configs/qm9_rebind.yaml
# so the two models are directly comparable.
#
# Reference numbers for QM9 test, as tabulated in the ReBind paper:
# GTMGC D-MAE 0.281 D-RMSE 0.471 C-RMSD 0.414
# ReBind D-MAE 0.254 D-RMSE 0.446 C-RMSD 0.321
#
# Architecture follows their released QM9 checkpoint (RichXuOvO/GTMGC-Qm9):
# 6 + 6 layers, d_model 256, d_ffn 1024, 8 heads, Mole-BERT token embeddings.
# Their conformer models embed tokenizer ids rather than atom types, so the
# per-atom ids must be precomputed first:
#
# uv run python scripts/tokenize_molebert.py -c configs/qm9_gtmgc.yaml \
# --out /home/gridsan/jtoney/step-up-data/qm9/qm9-molebert-tokens.csv
model: gtmgc
dataset_path: /home/gridsan/jtoney/step-up-data/qm9/qm9-rebind.csv
dataset_source: sdf
subset_size: null
id_column: mol_id
split_column: split
token_file: /home/gridsan/jtoney/step-up-data/qm9/qm9-molebert-tokens.csv
max_drop_fraction: 0.05

n_layers: 6
d_model: 256
d_ffn: 1024
n_head: 8
dropout: 0.0

# Optimization from their experiments/conformer_prediction/gtmgc_for_conformer_prediction.sh.
# That script is the Molecule3D one (the repo ships no QM9 variant), so the learning
# rate is theirs for Molecule3D; everything else is shared across their runs. As with
# ReBind we train in fp32 rather than their fp16 (see README).
epochs: 20
batch_size: 100
eval_batch_size: 100
lr: 5.0e-5
weight_decay: 0.0
warmup_ratio: 0.1
adam_beta1: 0.9
adam_beta2: 0.99
adam_eps: 1.0e-8
lr_schedule: linear
grad_clip: 1.0
drop_last: true
num_workers: 4

device: cuda
output_dir: outputs/qm9_gtmgc
seed: 42
44 changes: 44 additions & 0 deletions configs/qm9_rebind.yaml
Original file line number Diff line number Diff line change
@@ -0,0 +1,44 @@
# Reproduction run for ReBind's published QM9 conformer-prediction numbers
# (paper, test split: D-MAE 0.254, D-RMSE 0.446, C-RMSD 0.321).
#
# Data is the QM9 copy ReBind and GTMGC trained on (HuggingFace RichXuOvO/HFQm9),
# converted by scripts/prepare_qm9_rebind.py. Molblocks are copied out of gdb9.sdf
# verbatim, so bonds are the published ones rather than re-perceived from geometry,
# and the CSV carries their published split (110,000 / 10,000 / 10,831).
#
# Hyperparameters follow external/ReBIND/experiments/conformer_prediction/rebind.sh;
# see README for the few places the loop still differs from their script.
dataset_path: /home/gridsan/jtoney/step-up-data/qm9/qm9-rebind.csv
dataset_source: sdf
subset_size: null
id_column: mol_id
split_column: split
# About 1.4% of records fail RDKit sanitization and are dropped, as in ReBind's
# own evaluation; fail the run if that rate jumps.
max_drop_fraction: 0.05

n_layers: 8
d_model: 512
d_ffn: 1024
n_head: 8
dropout: 0.0

# Optimization, verbatim from rebind.sh (fp16 is the one setting we don't mirror;
# we train in fp32, which is the more precise choice — see README).
epochs: 20
batch_size: 100
eval_batch_size: 100
lr: 9.0e-5
weight_decay: 0.0
warmup_ratio: 0.1
adam_beta1: 0.9
adam_beta2: 0.99
adam_eps: 1.0e-8
lr_schedule: linear
grad_clip: 1.0
drop_last: true
num_workers: 4

device: cuda
output_dir: outputs/qm9_rebind
seed: 42
53 changes: 53 additions & 0 deletions configs/qm9_rebind_noh.yaml
Original file line number Diff line number Diff line change
@@ -0,0 +1,53 @@
# Heavy-atom QM9: the ReBind QM9 setup with hydrogens dropped from the graph.
# Not a reproduction of their published numbers — see the remove_hs note below.
# (paper, test split: D-MAE 0.254, D-RMSE 0.446, C-RMSD 0.321).
#
# Data is the QM9 copy ReBind and GTMGC trained on (HuggingFace RichXuOvO/HFQm9),
# converted by scripts/prepare_qm9_rebind.py. Molblocks are copied out of gdb9.sdf
# verbatim, so bonds are the published ones rather than re-perceived from geometry,
# and the CSV carries their published split (110,000 / 10,000 / 10,831).
#
# Hyperparameters follow external/ReBIND/experiments/conformer_prediction/rebind.sh;
# see README for the few places the loop still differs from their script.
dataset_path: /home/gridsan/jtoney/step-up-data/qm9/qm9-rebind.csv
dataset_source: sdf
subset_size: null
id_column: mol_id
split_column: split
# About 1.4% of records fail RDKit sanitization and are dropped, as in ReBind's
# own evaluation; fail the run if that rate jumps.
max_drop_fraction: 0.05

n_layers: 8
d_model: 512
d_ffn: 1024
n_head: 8
dropout: 0.0

# Optimization, verbatim from rebind.sh (fp16 is the one setting we don't mirror;
# we train in fp32, which is the more precise choice — see README).
epochs: 20
batch_size: 100
eval_batch_size: 100
lr: 9.0e-5
weight_decay: 0.0
warmup_ratio: 0.1
adam_beta1: 0.9
adam_beta2: 0.99
adam_eps: 1.0e-8
lr_schedule: linear
grad_clip: 1.0
drop_last: true
num_workers: 4

device: cuda
output_dir: outputs/qm9_rebind_noh
seed: 42

# Heavy-atom variant: hydrogens are dropped from the graph, so the model never
# sees or predicts them. Paired with configs/qm9_rebind.yaml to separate "the model is
# bad at hydrogen" from "the model is bad at geometry". Each heavy atom keeps its
# hydrogen count in the numH feature, so the graph is not missing the chemistry,
# only the positions. Note this departs from the published setup, which keeps
# hydrogen explicit, so it is not a reproduction run.
remove_hs: true
29 changes: 29 additions & 0 deletions configs/qm9_smoke.yaml
Original file line number Diff line number Diff line change
@@ -0,0 +1,29 @@
# Smallest possible run that exercises the full pipeline on CPU.
# Goal: prove the loss strictly decreases. Numbers are not meaningful.
dataset_path: /home/gridsan/jtoney/ElemeNet-benchmarking/benchmarking/datasets/QM9-full.csv
dataset_source: smiles
subset_size: 100
# Split key: mol_id, so QM9's few duplicated rows always share a split.
id_column: mol_id
cache_dataset: true
split_ratios: [0.8, 0.1, 0.1]
split_seed: 0

# Tiny model — enough capacity to learn distance regression on 80 toy molecules.
n_layers: 2
d_model: 64
d_ffn: 128
n_head: 4
dropout: 0.0

epochs: 3
batch_size: 8
eval_batch_size: 8
lr: 3.0e-4
weight_decay: 0.0
warmup_ratio: 0.1
num_workers: 0

device: cpu
output_dir: outputs/qm9_smoke
seed: 0
Loading
Loading