Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
55 changes: 55 additions & 0 deletions benchmarks/matbench_v0.1_Atomic_Affine_QRF/README.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,55 @@
# Atomic-AffQRF (matbench_v0.1, matbench_steels)

Composition-only regression on `matbench_steels`. Reference seed 417:
mean fold MAE 75.5426 MPa on the five official Matbench v0.1 folds.

## Method

- Features: atomic-fraction composition vectors under four fixed transforms.
- 16 fixed ExtraTrees configurations; each gives a leaf-distribution conditional
mean and lower-median prediction (32 predictors).
- Within each official outer training fold, five inner folds produce
out-of-fold (OOF) predictions for all 32 predictors.
- 128 Bayesian-bootstrap weighted-MAE linear programs fit nonnegative
coefficients (sum <= 2) plus a free intercept on the OOF predictions;
coefficients and intercepts are averaged. Outer-test labels are never used.
- Final forests (900 trees) are refit on the full outer training fold.

## Files

- `results.json.gz`, `info.json`, `notebook.ipynb`: required submission files.
- `run.py`, `inputs/base_qrf.py`: fitting source (base forests, OOF libraries,
affine LP aggregation). `inputs/protocol.json`, `protocol.json`: frozen settings.
- `inputs/matbench_steels.json.gz`, `inputs/steels_splits.json`: the official
dataset and fold indices used (snapshots of the Matbench release).
- `fresh.py`, `export.py`, `reproduce_entry.py`: end-to-end rebuild from raw
inputs and export to a Matbench results file.
- `preflight.py`, `environment_lock.json`, `manifest.json`,
`upstream_source_hashes.json`: refuse to run with mismatched package
versions, changed inputs, or pre-existing fit caches.
- `all_seed_scores.csv`: mean fold MAE for every run seed (0-6 and 417) and the
two ablation aggregators.

## Reproduce

```bash
uv venv --python 3.13.16 .venv
uv pip install --python .venv/bin/python -r requirements.txt
.venv/bin/python -B reproduce_entry.py --output fresh_script_review --workers 3
```

This rebuilds 400 inner forests and all active final forests from an empty fit
directory and asserts that the complete official fold dictionaries equal
`results.json.gz` exactly. The notebook runs the same pipeline.

## Limitations

- The official folds were used repeatedly during method development, so
model-family selection across runs may be optimistic even though each run
fits only on outer-training labels. This is not an untouched holdout.
- Seed repeats 0-6: 75.77 +/- 0.33 MPa (sample SD); the primary is the
pre-declared seed 417, not the best repeat.
- On a separate composition-grouped five-fold stress test (201 groups; not the
official task) seed 417 scores 99.56 MPa versus 99.09 MPa for the released
TPOT pipeline refit with the same seed, so no chemistry-transfer advantage
is claimed.
25 changes: 25 additions & 0 deletions benchmarks/matbench_v0.1_Atomic_Affine_QRF/all_seed_scores.csv
Original file line number Diff line number Diff line change
@@ -0,0 +1,25 @@
policy,seed,mean_fold_mae
bayesian_affine,417,75.54255828747938
bayesian_affine,0,75.88679088867481
bayesian_affine,1,75.1929192291949
bayesian_affine,2,75.52201809266612
bayesian_affine,3,76.16011643406503
bayesian_affine,4,76.06811188551217
bayesian_shift,417,75.68879242916975
bayesian_shift,0,76.13647443601975
bayesian_shift,1,75.65266598767167
bayesian_shift,2,75.82769487044612
bayesian_shift,3,76.39709826666487
bayesian_shift,4,76.29025043714486
bayesian_simplex,417,75.65145589673142
bayesian_simplex,0,76.13905197086122
bayesian_simplex,1,75.5751077961455
bayesian_simplex,2,75.81828399654795
bayesian_simplex,3,76.38059435407408
bayesian_simplex,4,76.34724882674907
bayesian_affine,5,75.85567005683806
bayesian_shift,5,76.0264297572666
bayesian_simplex,5,75.87422644721983
bayesian_affine,6,75.68486159911308
bayesian_shift,6,75.97393696973424
bayesian_simplex,6,75.98678770919275
81 changes: 81 additions & 0 deletions benchmarks/matbench_v0.1_Atomic_Affine_QRF/environment_lock.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,81 @@
{
"Pygments": "2.21.0",
"asttokens": "3.0.2",
"attrs": "26.1.0",
"bibtexparser": "1.4.4",
"certifi": "2026.7.22",
"charset-normalizer": "3.5.2",
"cloudpickle": "3.1.2",
"comm": "0.2.3",
"contourpy": "1.4.0",
"cycler": "0.12.1",
"debugpy": "1.8.22",
"dnspython": "2.8.0",
"executing": "2.2.1",
"fastjsonschema": "2.22.2",
"fonttools": "4.66.1",
"idna": "3.20",
"ipykernel": "7.4.0",
"ipython": "9.17.1",
"ipython_pygments_lexers": "1.1.1",
"jedi": "0.20.0",
"joblib": "1.6.0",
"jsonschema": "4.26.0",
"jsonschema-specifications": "2025.9.1",
"jupyter_client": "8.10.0",
"jupyter_core": "5.9.1",
"kiwisolver": "1.5.1",
"lxml": "6.1.3",
"matbench": "0.6",
"matminer": "0.10.1",
"matplotlib": "3.11.2",
"matplotlib-inline": "0.2.2",
"monty": "2026.7.16",
"mpmath": "1.3.0",
"narwhals": "2.26.0",
"nbclient": "0.11.0",
"nbformat": "5.11.1",
"nest-asyncio2": "1.7.3",
"networkx": "3.7",
"numpy": "2.5.3",
"orjson": "3.12.0",
"packaging": "26.3",
"palettable": "3.3.3",
"pandas": "2.3.3",
"parso": "0.8.7",
"pexpect": "4.9.0",
"pillow": "12.3.0",
"platformdirs": "4.12.3",
"plotly": "7.1.0",
"prompt_toolkit": "3.0.53",
"psutil": "7.2.2",
"ptyprocess": "0.7.0",
"pure_eval": "0.2.4",
"pymatgen": "2026.9.24",
"pymatgen-core": "2026.10.2",
"pymongo": "4.18.2",
"pyparsing": "3.3.3",
"python-dateutil": "2.9.0.post0",
"pytz": "2026.5",
"pyzmq": "27.2.0",
"referencing": "0.37.0",
"requests": "2.34.2",
"rpds-py": "2026.9.1",
"ruamel.yaml": "0.19.1",
"scikit-learn": "1.9.1",
"scipy": "1.18.1",
"six": "1.17.0",
"spglib": "2.7.0",
"stack-data": "0.6.3",
"sympy": "1.14.0",
"tabulate": "0.10.0",
"threadpoolctl": "3.7.0",
"tornado": "6.5.10",
"tqdm": "4.70.1",
"traitlets": "5.16.1",
"typing_extensions": "4.16.0",
"tzdata": "2026.5",
"uncertainties": "3.2.3",
"urllib3": "2.8.0",
"wcwidth": "0.9.2"
}
34 changes: 34 additions & 0 deletions benchmarks/matbench_v0.1_Atomic_Affine_QRF/export.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,34 @@
"""Export all frozen primary seeds to the official Matbench benchmark format."""
import gzip
import json
import numpy as np
import pandas as pd
from run import ROOT,PROTOCOL,base,dump
from matbench.bench import MatbenchBenchmark

def export_seed(seed,root=ROOT):
df,_,folds=base.load();folder=root/f'seed_{seed:03d}'
prediction=pd.concat([pd.read_csv(folder/f/'predictions.csv',float_precision='round_trip') for f in folds]).set_index('mbid')
mb=MatbenchBenchmark(autoload=False,subset=['matbench_steels']);task=mb.matbench_steels
task.df=df[['composition','yield strength']]
for fi in task.folds:
f=f'fold_{fi}';test=task.get_test_data(fi,include_target=False)
assert test.index.tolist()==df.index[folds[f][1]].tolist()
fit=json.loads((folder/f/'weights.json').read_text())[PROTOCOL['primary']]
task.record(fi,prediction.loc[test.index,PROTOCOL['primary']].to_numpy(),params=dict(
seed=seed,**fit,inner_trees=600,final_trees=900,inner_folds=5,bootstrap_draws=128,
bootstrap_seed=19072026,coefficient_sum_bounds=[0.,2.],selection='Nonnegative affine MAE fits on 128 Bayesian bootstrap samples of outer-training OOF predictions'))
mb.user_metadata=dict(algorithm=PROTOCOL['name'],protocol=PROTOCOL,verified=False,
application_scope='Retrospective composition benchmark; no prospective or physical qualification claim.')
mb.validate();path=folder/'results.json.gz'
path.write_bytes(gzip.compress(json.dumps(mb.as_dict()).encode(),mtime=0))
reread=MatbenchBenchmark.from_dict(json.loads(gzip.decompress(path.read_bytes())));reread.validate()
expected=np.mean([np.mean(abs(prediction.loc[df.index[te],PROTOCOL['primary']].to_numpy()-df.iloc[te]['yield strength'].to_numpy())) for _,te in folds.values()])
assert abs(expected-reread.matbench_steels.scores.mae.mean)<1e-10
return dict(seed=seed,mean_fold_mae=float(expected),passed=True)

def main():
records=[export_seed(s) for s in [417,*PROTOCOL['repeats']]]
dump(ROOT/'official_exports.json',records);print(json.dumps(records,indent=2))

if __name__=='__main__':main()
64 changes: 64 additions & 0 deletions benchmarks/matbench_v0.1_Atomic_Affine_QRF/fresh.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,64 @@
"""One bounded fresh replay, starting from chemistry and official split files only."""
import os
for key in ('OMP_NUM_THREADS','OPENBLAS_NUM_THREADS','MKL_NUM_THREADS','NUMEXPR_NUM_THREADS'):
os.environ[key]='1'
import argparse
from concurrent.futures import ProcessPoolExecutor
import json
from pathlib import Path
import shutil
import time
import numpy as np
import pandas as pd
import run

ROOT=Path(__file__).resolve().parent

def fold_from_scratch(stage,seed,fold):
run.ROOT=Path(stage);run.base.ROOT=run.ROOT/'inputs'
# Rebuild all sixteen five-fold OOF libraries, using only training rows.
run.base.run_fold(seed,fold)
df,_,folds=run.base.load();tr,te=folds[fold]
path=run.ROOT/'inputs'/f'seed_{seed:03d}'/fold/'final_components.npz'
np.savez_compressed(path,components=np.full((len(te),16,2),np.nan),
active_configs=np.array([],dtype=int),train_indices=tr,test_indices=te)
return run.run_fold(seed,fold)

def reproduce(stage,seed=417,workers=3,compare_reference=True):
stage=Path(stage).resolve();assert stage.is_relative_to(ROOT),stage
assert not stage.exists(),'Fresh replay refuses an existing directory.'
assert 1<=workers<=3
start=time.monotonic();(stage/'inputs').mkdir(parents=True)
for f in ['base_qrf.py','protocol.json','matbench_steels.json.gz','steels_splits.json']:
shutil.copyfile(ROOT/'inputs'/f,stage/'inputs'/f)
with ProcessPoolExecutor(workers) as pool:
tasks=[pool.submit(fold_from_scratch,str(stage),seed,f'fold_{f}') for f in range(5)]
results=[t.result() for t in tasks]
# Full exact new-implementation replay; no previously fit arrays reused.
maximum=0.
for f in range(5) if compare_reference else []:
fold=f'fold_{f}';original=ROOT/f'seed_{seed:03d}'/fold;fresh=stage/f'seed_{seed:03d}'/fold
assert json.loads((fresh/'weights.json').read_text())==json.loads((original/'weights.json').read_text())
oo=np.load(stage/'inputs'/f'seed_{seed:03d}'/fold/'inner_predictions.npz')
cached=np.load(ROOT/'inputs'/f'seed_{seed:03d}'/fold/'inner_predictions.npz')
np.testing.assert_array_equal(oo['oof'],cached['oof'])
a=pd.read_csv(original/'predictions.csv',float_precision='round_trip')
b=pd.read_csv(fresh/'predictions.csv',float_precision='round_trip')
assert a.mbid.tolist()==b.mbid.tolist()
for p in run.POLICIES:
difference=float(np.max(abs(a[p]-b[p])));maximum=max(maximum,difference)
np.testing.assert_array_equal(a[p],b[p])
result=dict(passed=True,seed=seed,empty_fit_directory_at_start=True,reused_training_cache=False,
inner_forests=400,auxiliary_old_selector_forests=5,
active_final_forests=sum(len(r['active_configs']) for r in results),
max_prediction_difference_mpa=maximum if compare_reference else None,seconds=time.monotonic()-start,
exact_weights_and_inner_predictions=compare_reference,verified=False)
run.dump(stage/'fresh_verification.json',result);print(json.dumps(result,indent=2),flush=True)
return result

def main():
ap=argparse.ArgumentParser();ap.add_argument('--output',default='fresh_reference');ap.add_argument('--seed',type=int,default=417)
ap.add_argument('--workers',type=int,default=3);a=ap.parse_args()
reproduce(ROOT/a.output,a.seed,a.workers)

if __name__=='__main__':main()
95 changes: 95 additions & 0 deletions benchmarks/matbench_v0.1_Atomic_Affine_QRF/info.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,95 @@
{
"authors": "Touko Ursin",
"algorithm": "Atomic-AffQRF",
"algorithm_long": "Atomic-Affine-Bagged-QRF: Bayesian-bootstrap nonnegative affine MAE combination of 32 ExtraTrees conditional mean/median predictors, with coefficient sum at most two and free intercept. Five training-only inner folds; fixed 16 configurations, 600 inner trees and 900 final trees; reference seed 417.",
"bibtex_refs": [
"@article{meinshausen2006quantile, author={Nicolai Meinshausen}, title={Quantile Regression Forests}, journal={Journal of Machine Learning Research}, volume={7}, pages={983--999}, year={2006}, url={https://jmlr.org/papers/v7/meinshausen06a.html}}",
"@article{geurts2006extremely, author={Pierre Geurts and Damien Ernst and Louis Wehenkel}, title={Extremely randomized trees}, journal={Machine Learning}, volume={63}, pages={3--42}, year={2006}}",
"@article{rubin1981bayesian, author={Donald B. Rubin}, title={The Bayesian Bootstrap}, journal={The Annals of Statistics}, volume={9}, pages={130--134}, year={1981}, doi={10.1214/aos/1176345338}}",
"@article{dunn2020matbench, author={Alexander Dunn and Qi Wang and Alex Ganose and Daniel Dopp and Anubhav Jain}, title={Benchmarking materials property prediction methods: the Matbench test set and Automatminer reference algorithm}, journal={npj Computational Materials}, volume={6}, year={2020}, doi={10.1038/s41524-020-00406-3}}"
],
"notes": "Composition-only. Reference seed 417 reported; seed repeats 0-6 give 75.77 +/- 0.33 MPa mean fold MAE (all in all_seed_scores.csv). The official folds were reused during method development, so selection across runs may be optimistic; each run fits only on outer-training labels. Exact rebuild requires CPython 3.13.16 and requirements.txt; reproduce_entry.py and the notebook rebuild all forests from raw inputs and check the fold dictionaries against results.json.gz exactly.",
"requirements": {
"python": [
"asttokens==3.0.2",
"attrs==26.1.0",
"bibtexparser==1.4.4",
"certifi==2026.7.22",
"charset-normalizer==3.5.2",
"cloudpickle==3.1.2",
"comm==0.2.3",
"contourpy==1.4.0",
"cycler==0.12.1",
"debugpy==1.8.22",
"dnspython==2.8.0",
"executing==2.2.1",
"fastjsonschema==2.22.2",
"fonttools==4.66.1",
"idna==3.20",
"ipykernel==7.4.0",
"ipython==9.17.1",
"ipython-pygments-lexers==1.1.1",
"jedi==0.20.0",
"joblib==1.6.0",
"jsonschema==4.26.0",
"jsonschema-specifications==2025.9.1",
"jupyter-client==8.10.0",
"jupyter-core==5.9.1",
"kiwisolver==1.5.1",
"lxml==6.1.3",
"matbench @ https://github.com/materialsproject/matbench/archive/936176db18ca4cd7b38cbd957c017a5bac770c6b.zip",
"matminer==0.10.1",
"matplotlib==3.11.2",
"matplotlib-inline==0.2.2",
"monty==2026.7.16",
"mpmath==1.3.0",
"narwhals==2.26.0",
"nbclient==0.11.0",
"nbformat==5.11.1",
"nest-asyncio2==1.7.3",
"networkx==3.7",
"numpy==2.5.3",
"orjson==3.12.0",
"packaging==26.3",
"palettable==3.3.3",
"pandas==2.3.3",
"parso==0.8.7",
"pexpect==4.9.0",
"pillow==12.3.0",
"platformdirs==4.12.3",
"plotly==7.1.0",
"prompt-toolkit==3.0.53",
"psutil==7.2.2",
"ptyprocess==0.7.0",
"pure-eval==0.2.4",
"pygments==2.21.0",
"pymatgen==2026.9.24",
"pymatgen-core==2026.10.2",
"pymongo==4.18.2",
"pyparsing==3.3.3",
"python-dateutil==2.9.0.post0",
"pytz==2026.5",
"pyzmq==27.2.0",
"referencing==0.37.0",
"requests==2.34.2",
"rpds-py==2026.9.1",
"ruamel-yaml==0.19.1",
"scikit-learn==1.9.1",
"scipy==1.18.1",
"six==1.17.0",
"spglib==2.7.0",
"stack-data==0.6.3",
"sympy==1.14.0",
"tabulate==0.10.0",
"threadpoolctl==3.7.0",
"tornado==6.5.10",
"tqdm==4.70.1",
"traitlets==5.16.1",
"typing-extensions==4.16.0",
"tzdata==2026.5",
"uncertainties==3.2.3",
"urllib3==2.8.0",
"wcwidth==0.9.2"
]
}
}
Loading
Loading