"""Writes the deposit the paper requires before training: which compounds are on each side of each split, the reference value of every held-out compound, and the predictions of the baselines for it. Input: data/alexandria_ph/dataset.pkl and model/data/labels.csv (written by model/dataset.py) Output: model/deposit/splits.csv one row per compound with its side in every split model/deposit/baselines.csv baseline predictions of h for every held-out hydride model/deposit/manifest.json what was deposited, when, and the checksums model/train.py reads its splits from this deposit and refuses to run without it. Run: .venv/bin/python model/deposit.py """ import csv import hashlib import json import math import pathlib import subprocess import sys from datetime import datetime, timezone import numpy as np sys.path.insert(0, str(pathlib.Path(__file__).resolve().parent)) import train as T ROOT = pathlib.Path(__file__).resolve().parents[1] OUT = ROOT / 'model' / 'deposit' FOLDS = 5 def sha(path): return hashlib.sha256(open(path, 'rb').read()).hexdigest() def main(): rows = T.load() OUT.mkdir(exist_ok=True) held_top = T.split_top(rows) protos = sorted({r['row']['prototype'] for r in rows}) rng = np.random.default_rng(0) rng.shuffle(protos) grouped = {p: i % FOLDS for i, p in enumerate(protos)} perm = np.random.default_rng(0).permutation(len(rows)) random_fold = {rows[q]['row']['id']: i % FOLDS for i, q in enumerate(perm)} with open(OUT / 'splits.csv', 'w', newline='') as f: w = csv.writer(f) w.writerow(['id', 'formula', 'prototype', 'has_h', 'h_reference', 'S_reference', 'top', 'a2mh6', 'grouped_fold', 'random_fold']) for r in rows: x = r['row'] w.writerow([x['id'], x['formula'], x['prototype'], int(r['has_h']), '%.6g' % x['h'] if r['has_h'] else '', '%.6g' % x['S'], 'test' if x['prototype'] in held_top else 'train', 'test' if x['prototype'] == T.A2MH6 else 'train', grouped[x['prototype']], random_fold[x['id']]]) def sides(name, is_test): test = [r for r in rows if is_test(r) and r['has_h']] train_h = [r for r in rows if not is_test(r) and r['has_h']] return name, train_h, test tests = [sides('top', lambda r: r['row']['prototype'] in held_top), sides('a2mh6', lambda r: r['row']['prototype'] == T.A2MH6)] for k in range(FOLDS): tests.append(sides(f'grouped_{k}', lambda r, k=k: grouped[r['row']['prototype']] == k)) tests.append(sides(f'random_{k}', lambda r, k=k: random_fold[r['row']['id']] == k)) counts = {} with open(OUT / 'baselines.csv', 'w', newline='') as f: w = csv.writer(f) w.writerow(['split', 'id', 'h_reference', 'h_gas', 'h_dos', 'h_const']) for name, train_h, test in tests: counts[name] = dict(train_hydrides=len(train_h), test_hydrides=len(test), test_prototypes=len({r['row']['prototype'] for r in test})) if not test: continue b = T.baselines(train_h, test) for i, r in enumerate(test): w.writerow([name, r['row']['id'], '%.6g' % r['row']['h']] + ['%.6g' % math.exp(b[k][i]) for k in ('gas', 'dos', 'const')]) commit = subprocess.run(['git', 'rev-parse', 'HEAD'], cwd=ROOT, capture_output=True, text=True).stdout.strip() manifest = dict( deposited_utc=datetime.now(timezone.utc).strftime('%Y-%m-%dT%H:%M:%SZ'), code_commit_before_deposit=commit, source='Alexandria phonon and electron-phonon release of 2025-08-11, PBEsol, 3D, 93 files, CC BY 4.0', label='h = S_top / rho_H at a smearing of 0.030 Ry, for hydrides whose 3 n_H highest branches each carry at least 90% of their eigenvector weight on hydrogen at every stored wavevector', compounds_with_spectrum=len(rows), hydrides_with_h=sum(r['has_h'] for r in rows), prototypes=len(protos), splits=dict( top='prototypes ranked by their largest h; the leading ones are held out with all their members until a tenth of the labelled hydrides are held out', a2mh6='every compound of prototype %s (space group 225, cubic A2MH6) is held out' % T.A2MH6, grouped='five folds of whole prototypes, shuffled with seed 0', random='five folds of compounds, shuffled with seed 0; reported without a mark'), counts=counts, baselines=dict( gas='36.6 eV A for every compound', dos='a constant fitted on the training side times the total density of states at the Fermi level per unit volume. The paper specifies the hydrogen-projected density of states, which the release does not store', const='the geometric mean of h on the training side'), mark='mean absolute error in ln h at most half that of the best baseline, with the 95% interval of the ratio, from resampling held-out prototypes, below one', amendments=[ 'The training-side geometric mean is added as a third baseline and the estimator is judged against the best of the three. The paper fixed two baselines. A code test on nine of the 93 files, run on 2026-10-06 before this deposit, showed both to be far from the labels (mean absolute errors of 1.80 and 1.11 in ln h, against 0.66 for the training mean), so a mark against them alone would be met without skill. The result against the two original baselines is reported as well.', 'Two code tests on those nine files preceded this deposit: both estimators on five grouped folds, 15 epochs for the graph network (output kept in model/results/smoke/), and the hold-out-the-top split with 2 epochs (output discarded). No estimator had been trained on the full release when this deposit was written.'], not_run=[ 'Hold out pressure and megabar to one atmosphere: the release holds no calculation at megabar pressure.', 'Calibration against a reference set of dense calculations: no such set exists yet.', 'Closure of the spectrum and the end-to-end check on the transition temperature: not part of this deposit.'], checksums={p.name: sha(p) for p in (ROOT / 'model' / 'data' / 'labels.csv', OUT / 'splits.csv', OUT / 'baselines.csv')}) json.dump(manifest, open(OUT / 'manifest.json', 'w'), indent=1) print(json.dumps({k: manifest[k] for k in ('deposited_utc', 'compounds_with_spectrum', 'hydrides_with_h', 'prototypes', 'counts')}, indent=1)) if __name__ == '__main__': main()