Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
107 changes: 107 additions & 0 deletions benchmarks/matbench_v0.1_alignn_v2.0/collect_matbench.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,107 @@
"""Record trained-fold predictions into a Matbench results file.

Reads each fold's prediction_results_test_set.csv (id,target,prediction;
ids carry .vasp) from --preds_root/mb_run_<task>_f<fold>/, maps ids back
to matbench test order via the local fold_manifest.json written by
make_matbench_dataset.py, records with task.record, prints per-fold and
task scores, and writes results.json.gz.

Run locally (needs matbench). Tasks with any missing fold are skipped
with a warning, so partial collections work during the campaign.
"""
import argparse
import csv
import json
import os

import pandas as pd
from matbench.bench import MatbenchBenchmark

CLF_TASKS = {"matbench_mp_is_metal"}


def short(task):
"""matbench_mp_e_form -> mp_e_form."""
return task.replace("matbench_", "")


def load_preds(path):
"""prediction csv -> {mbid: float}."""
out = {}
with open(path) as f:
for row in csv.reader(f):
if row[0] == "id":
continue
out[row[0].replace(".vasp", "")] = float(row[2])
return out


def main():
"""Record all requested tasks' folds and write results.json.gz."""
ap = argparse.ArgumentParser()
ap.add_argument("--tasks", nargs="+", required=True)
ap.add_argument("--preds_root", default=".",
help="contains mb_run_<task>_f<fold>/ dirs")
ap.add_argument("--data_root", default=".",
help="contains mb_<task>_f<fold>/fold_manifest.json")
ap.add_argument("--out", default="results.json.gz")
args = ap.parse_args()

mb = MatbenchBenchmark(autoload=False, subset=args.tasks)
for task in mb.tasks:
name = task.dataset_name
task.load()
complete = True
for fold in task.folds:
tag = "%s_f%d" % (short(name), fold)
pred_csv = os.path.join(args.preds_root, "mb_run_" + tag,
"prediction_results_test_set.csv")
manifest = os.path.join(args.data_root, "mb_" + tag,
"fold_manifest.json")
if not (os.path.exists(pred_csv) and os.path.exists(manifest)):
print("MISSING fold:", tag, "- skipping task", name)
complete = False
break
test_ids = json.load(open(manifest))["test_ids"]
# Classification: matbench 0.6 computes every clf metric from
# labels (its rocauc branch sees a pred_array the accuracy branch
# already rebound to labels), so recording probabilities gains
# nothing and the decision threshold is the only knob. Use the
# validation-selected threshold when matbench_clf_probs.py wrote
# one, else fall back to the 0.5 labels train.py produced.
threshold = 0.5
if name in CLF_TASKS:
run_dir = os.path.join(args.preds_root, "mb_run_" + tag)
prob_csv = os.path.join(run_dir,
"prediction_probs_test_set.csv")
thr_json = os.path.join(run_dir, "clf_threshold.json")
if os.path.exists(prob_csv):
pred_csv = prob_csv
if os.path.exists(thr_json):
with open(thr_json) as fh:
threshold = json.load(fh)["threshold"]
print(" %s: probabilities, threshold %.2f"
% (tag, threshold))
preds = load_preds(pred_csv)
missing = [i for i in test_ids if i not in preds]
assert not missing, "%s: %d test ids without predictions" % (
tag, len(missing))
vals = [preds[i] for i in test_ids]
if name in CLF_TASKS:
vals = [v > threshold for v in vals]
task.record(fold, pd.Series(vals, index=test_ids))
fold_key = "fold_%d" % fold
fold_scores = task.results.get(fold_key, {}).get("scores", {})
print("recorded", tag, "| fold scores:",
json.dumps(dict(fold_scores), default=str))
if complete:
print(name, "task scores:",
json.dumps(task.scores, default=str))

mb.to_file(args.out)
print("wrote", args.out, "| complete:", mb.is_complete,
"| valid:", mb.is_valid)


if __name__ == "__main__":
main()
94 changes: 94 additions & 0 deletions benchmarks/matbench_v0.1_alignn_v2.0/config_mb_base.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,94 @@
{
"version": "NA",
"dataset": "user_data",
"target": "target",
"atom_features": "cgcnn",
"neighbor_strategy": "pure_torch",
"id_tag": "jid",
"dtype": "float32",
"random_seed": 123,
"classification_threshold": null,
"target_multiplication_factor": null,
"epochs": 300,
"batch_size": 64,
"gpu_memory_fraction": null,
"use_amp": false,
"ddp_find_unused_parameters": false,
"resume_checkpoint": false,
"lr_total_epochs": null,
"deterministic": false,
"weight_decay": 1e-05,
"learning_rate": 0.001,
"filename": "sample",
"warmup_steps": 2000,
"criterion": "l1",
"optimizer": "adamw",
"scheduler": "onecycle",
"pin_memory": false,
"save_dataloader": false,
"write_checkpoint": true,
"write_predictions": true,
"store_outputs": true,
"progress": true,
"log_tensorboard": false,
"standard_scalar_and_pca": false,
"use_canonize": true,
"compute_line_graph": true,
"num_workers": 2,
"cutoff": 8.0,
"cutoff_extra": 3.0,
"three_body_cutoff": 8.0,
"max_neighbors": 12,
"keep_data_order": true,
"normalize_graph_level_loss": false,
"distributed": false,
"data_parallel": false,
"n_early_stopping": null,
"eval_best_checkpoint": true,
"use_ema": false,
"ema_decay": 0.999,
"torch_seed": 123,
"output_dir": "mb_run_TASK_fFOLD",
"use_lmdb": true,
"read_existing": false,
"model": {
"name": "alignn_atomwise_pure",
"alignn_layers": 4,
"gcn_layers": 4,
"atom_input_features": 92,
"edge_input_features": 80,
"triplet_input_features": 40,
"embedding_features": 64,
"hidden_features": 768,
"output_features": 1,
"grad_multiplier": -1,
"calculate_gradient": false,
"atomwise_output_features": 0,
"graphwise_weight": 1.0,
"gradwise_weight": 0.0,
"stresswise_weight": 0.0,
"atomwise_weight": 0.0,
"link": "identity",
"zero_inflated": false,
"classification": false,
"force_mult_natoms": false,
"energy_mult_natoms": true,
"include_pos_deriv": false,
"use_cutoff_function": false,
"inner_cutoff": 3.0,
"stress_multiplier": 1.0,
"add_reverse_forces": true,
"lg_on_fly": true,
"batch_stress": true,
"multiply_cutoff": false,
"use_penalty": true,
"extra_features": 0,
"exponent": 5,
"penalty_factor": 0.5,
"penalty_threshold": 1.0,
"additional_output_features": 0,
"additional_output_weight": 0.0,
"conv_type": "alignn",
"num_heads": 4
}
}
Original file line number Diff line number Diff line change
@@ -0,0 +1,97 @@
{
"version": "NA",
"dataset": "user_data",
"target": "target",
"atom_features": "cgcnn",
"neighbor_strategy": "pure_torch",
"id_tag": "jid",
"dtype": "float32",
"random_seed": 123,
"classification_threshold": 0.5,
"target_multiplication_factor": null,
"epochs": 75,
"batch_size": 64,
"gpu_memory_fraction": null,
"use_amp": false,
"ddp_find_unused_parameters": false,
"resume_checkpoint": false,
"lr_total_epochs": null,
"deterministic": false,
"weight_decay": 1e-05,
"learning_rate": 0.001,
"filename": "sample",
"warmup_steps": 2000,
"criterion": "l1",
"optimizer": "adamw",
"scheduler": "onecycle",
"pin_memory": false,
"save_dataloader": false,
"write_checkpoint": true,
"write_predictions": true,
"store_outputs": true,
"progress": true,
"log_tensorboard": false,
"standard_scalar_and_pca": false,
"use_canonize": true,
"compute_line_graph": true,
"num_workers": 2,
"cutoff": 8.0,
"cutoff_extra": 3.0,
"three_body_cutoff": 8.0,
"max_neighbors": 12,
"keep_data_order": true,
"normalize_graph_level_loss": false,
"distributed": false,
"data_parallel": false,
"n_early_stopping": null,
"eval_best_checkpoint": true,
"use_ema": false,
"ema_decay": 0.999,
"torch_seed": 123,
"output_dir": "mb_run_mp_is_metal_f0",
"use_lmdb": true,
"read_existing": false,
"model": {
"name": "alignn_atomwise_pure",
"alignn_layers": 4,
"gcn_layers": 4,
"atom_input_features": 92,
"edge_input_features": 80,
"triplet_input_features": 40,
"embedding_features": 64,
"hidden_features": 768,
"output_features": 1,
"grad_multiplier": -1,
"calculate_gradient": false,
"atomwise_output_features": 0,
"graphwise_weight": 1.0,
"gradwise_weight": 0.0,
"stresswise_weight": 0.0,
"atomwise_weight": 0.0,
"link": "identity",
"zero_inflated": false,
"classification": true,
"force_mult_natoms": false,
"energy_mult_natoms": false,
"include_pos_deriv": false,
"use_cutoff_function": false,
"inner_cutoff": 3.0,
"stress_multiplier": 1.0,
"add_reverse_forces": true,
"lg_on_fly": true,
"batch_stress": true,
"multiply_cutoff": false,
"use_penalty": true,
"extra_features": 0,
"exponent": 5,
"penalty_factor": 0.5,
"penalty_threshold": 1.0,
"additional_output_features": 0,
"additional_output_weight": 0.0,
"conv_type": "alignn",
"num_heads": 4
},
"n_train": 80646,
"n_val": 4244,
"n_test": 21223
}
19 changes: 19 additions & 0 deletions benchmarks/matbench_v0.1_alignn_v2.0/info.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,19 @@
{
"authors": "Jaehyung Lee, Kamal Choudhary",
"algorithm": "ALIGNN 2.0",
"algorithm_long": "ALIGNN 2.0 is a dependency-free, pure-PyTorch reimplementation of the Atomistic Line Graph Neural Network (ALIGNN). Each crystal is converted into an atom graph and its line graph (bond-angle graph); four edge-gated ALIGNN layers alternate between the two graphs and are followed by four graph convolutions on the atom graph, and the mean-pooled representation feeds a single linear output head. The k-nearest-neighbour graph (12 neighbours within 8 A, full line graph) is used for all tasks, as recommended for property prediction in the ALIGNN 2.0 paper. All nine structure-based Matbench v0.1 tasks (jdft2d, phonons, dielectric, log_gvrh, log_kvrh, perovskites, mp_gap, mp_is_metal, mp_e_form) were run with ONE fixed configuration chosen before the benchmark, from earlier work on the JARVIS-Leaderboard: hidden dimension 768, 4 ALIGNN + 4 GCN layers, batch size 64, AdamW with weight decay 1e-5, one-cycle learning-rate schedule with peak 1e-3, L1 loss, no exponential moving average, random seed 123. The only per-task setting is the number of epochs, fixed a priori as a compute budget by dataset size (300 epochs for tasks under 5k samples, 200 for 5k-20k, 75 for over 100k) and identical for all folds of a task; it was never tuned on performance. For each of the five official folds, 5% of the training fold (a seeded random slice) is held out as a validation set used only to select the best checkpoint; the test fold is never touched during training or model selection. mp_is_metal is trained as a two-class classifier (negative log-likelihood loss, 2-logit head) and predictions are hard labels at the default 0.5 decision threshold; no threshold tuning was applied. The results.json.gz was produced with matbench's own MatbenchBenchmark API (task.record on each fold).",
"bibtex_refs": "@article{lee2026alignn2,\n title = {ALIGNN 2.0: A Unified Line-Graph Neural Network Framework for Materials Screening, Force Fields, Inverse Design, Spectroscopy, and Microscopy},\n author = {Lee, Jaehyung and Campbell, Charles Rhys and Ajith, Akshaya and Kalinin, Sergei V. and Wolverton, Christopher and Choudhary, Kamal},\n journal = {arXiv preprint arXiv:2609.19487},\n year = {2026},\n url = {https://arxiv.org/abs/2609.19487}\n}\n@article{choudhary2021alignn,\n title = {Atomistic Line Graph Neural Network for improved materials property predictions},\n author = {Choudhary, Kamal and DeCost, Brian},\n journal = {npj Computational Materials},\n volume = {7},\n number = {1},\n pages = {185},\n year = {2021},\n doi = {10.1038/s41524-021-00650-1}\n}\n@article{choudhary2024leaderboard,\n title = {JARVIS-Leaderboard: a large scale benchmark of materials design methods},\n author = {Choudhary, Kamal and Wines, Daniel and Li, Kang and Garrity, Kevin F and others},\n journal = {npj Computational Materials},\n volume = {10},\n number = {1},\n pages = {93},\n year = {2024}\n}",
"notes": "Structure-based tasks only (9 of 13); the four composition-only tasks are not applicable, since ALIGNN requires a crystal structure as input. Code: https://github.com/atomgptlab/alignn (pure-torch model alignn_atomwise_pure, neighbor_strategy pure_torch). Data pipeline: make_matbench_dataset.py pulls each fold through the matbench API (task.get_train_and_val_data / task.get_test_data), converts pymatgen structures to POSCAR files, and writes a folder-mode dataset with a fixed train/val/test order recorded in fold_manifest.json; the per-fold config locks n_train/n_val/n_test to that order (keep_data_order=true). Training: alignn.train_alignn on one NVIDIA GB10 GPU per fold; measured wall time per fold: 0.4 h for jdft2d, 0.7 h for phonons, 3.6-4.4 h for dielectric, log_gvrh, log_kvrh and perovskites, 42 h for mp_gap and mp_is_metal, and 50-51 h for mp_e_form. All reported folds were trained on the same cluster (JHU atomgptlab, GB10 nodes). As a reproducibility check the five mp_gap folds and the five mp_is_metal folds were additionally retrained on a second cluster (JHU Skipjack, NVIDIA A100/H100) with identical data and configuration, using a newer snapshot of the same ALIGNN source; for mp_gap the two runs of a fold differed by up to 1.5% in test MAE (+1.4, +1.5, +0.2, +0.0, -0.4%, second cluster minus first), with no consistent direction, i.e. run-to-run variation between the two setups comparable to the fold-to-fold spread. For mp_is_metal the second cluster gave a mean ROCAUC of 0.9032 against the reported 0.9042, differing fold by fold by -0.2 to +0.4%, again with no consistent direction; ALIGNN 2.0 scores below the original ALIGNN (0.9128) on this task on both clusters. Collection: collect_matbench.py maps each fold's predictions back to matbench test order via fold_manifest.json and records them with task.record. Classification note: for mp_is_metal the model config sets energy_mult_natoms=false (the per-atom energy scaling used for regression targets is not meaningful for class logits); everything else is identical to the regression configuration. matbench's classifier metrics were computed from the hard 0/1 predictions.",
"requirements": {
"python": [
"alignn==2026.5.20 (run from a source checkout of the pure-PyTorch development branch; repository linked in the notes)",
"torch==2.13.0+cu130",
"numpy==2.2.6",
"pydantic==2.13.4",
"jarvis-tools==2026.6.12",
"pymatgen==2025.10.7",
"matminer==0.9.3",
"matbench==0.6"
]
}
}
Loading
Loading