"""Runs the tests of the second deposit (model/deposit2/). Same estimators, baselines, mark and settings as model/train.py; the splits come from the deposit, which labels structure types independently of the origin and keeps one record per compound. Run: .venv/bin/python model/train2.py [epochs] Output: model/results2/.json and .log """ import csv import hashlib import json import math import pathlib import sys import time import numpy as np sys.path.insert(0, str(pathlib.Path(__file__).resolve().parent)) import train as T ROOT = T.ROOT DEP = ROOT / 'model' / 'deposit2' OUT = ROOT / 'model' / 'results2' def deposit(): man = json.load(open(DEP / 'manifest.json')) for name, want in man['checksums'].items(): p = (DEP / name) if (DEP / name).exists() else (ROOT / 'model' / 'data' / name) if hashlib.sha256(open(p, 'rb').read()).hexdigest() != want: raise SystemExit(f'{name} differs from the deposited one') return {r['id']: r for r in csv.DictReader(open(DEP / 'splits.csv'))}, man def run(train_all, test_all, name, log, epochs, seed): train_h = [r for r in train_all if r['has_h']] test_h = [r for r in test_all if r['has_h']] log(f'{name}: train {len(train_all)} compounds ({len(train_h)} hydrides with h), test {len(test_all)} compounds ({len(test_h)} hydrides with h)') preds = T.baselines(train_h, test_h) preds['trees'], _ = T.fit_trees(train_h, test_h, 'h') t = time.time() net = T.fit_graph(train_all, epochs=epochs, seed=seed, log=log) ln_s, _ = T.predict_graph(net, test_all) _, preds['graph'] = T.predict_graph(net, test_h) log(f' graph trained in {time.time() - t:.0f} s') s_const = np.mean([math.log(r['row']['S']) for r in train_all]) return train_h, test_h, preds, ln_s, s_const def score_s(test_all, ln_s, s_const, groups): y = np.array([math.log(r['row']['S']) for r in test_all]) e_g, e_c = np.abs(ln_s - y), np.abs(s_const - y) rng = np.random.default_rng(0) ug = np.unique(groups) idx = {g: np.flatnonzero(groups == g) for g in ug} ratios = [] for _ in range(2000): d = np.concatenate([idx[g] for g in rng.choice(ug, len(ug))]) ratios.append(e_g[d].mean() / e_c[d].mean()) return dict(n=int(len(y)), mae_graph=float(e_g.mean()), mae_training_mean=float(e_c.mean()), ratio=float(e_g.mean() / e_c.mean()), interval=[float(np.percentile(ratios, 2.5)), float(np.percentile(ratios, 97.5))], spearman=float(T.spearman(ln_s, y))) def main(): exp = sys.argv[1] epochs = int(sys.argv[2]) if len(sys.argv) > 2 else 60 OUT.mkdir(exist_ok=True) logf = open(OUT / f'{exp}.log', 'a') def log(s): print(s, flush=True) logf.write(s + '\n'); logf.flush() dep, man = deposit() rows = [] for r in T.load(): d = dep[r['row']['id']] if d['top'] == 'drop': continue r['dep'] = d r['stype'] = d['structure_type'] rows.append(r) log(f'== {exp}: {len(rows)} compounds kept, {sum(r["has_h"] for r in rows)} hydrides with h; deposit of {man["deposited_utc"]}') res = dict(experiment=exp, epochs=epochs, deposit=man['deposited_utc'], compounds=len(rows)) def pack(test_h, y, preds): return [dict(id=r['row']['id'], formula=r['row']['formula'], structure_type=r['stype'], h=r['row']['h'], **{k: float(math.exp(p[i])) for k, p in preds.items()}) for i, r in enumerate(test_h)] if exp in ('top', 'family'): tr = [r for r in rows if r['dep'][exp] == 'train'] te = [r for r in rows if r['dep'][exp] == 'test'] train_h, test_h, preds, ln_s, s_const = run(tr, te, exp, log, epochs, 0) y = np.array([math.log(r['row']['h']) for r in test_h]) groups = np.array([r['stype'] if exp == 'top' else r['row']['id'] for r in test_h]) res['score'] = T.score(y, preds, groups) res['S'] = score_s(te, ln_s, s_const, np.array([r['stype'] if exp == 'top' else r['row']['id'] for r in te])) res['compounds_held_out'] = pack(test_h, y, preds) if exp == 'top': ceiling = max(math.log(r['row']['h']) for r in train_h) above = y > ceiling res['training_maximum_h'] = math.exp(ceiling) res['held_out_above_training_maximum'] = int(above.sum()) for k in ('trees', 'graph'): res[k + '_extrapolation'] = dict(predicted_above_among_those_above=float((preds[k][above] > ceiling).mean()) if above.any() else None, predicted_above_among_the_rest=float((preds[k][~above] > ceiling).mean()) if (~above).any() else None) elif exp in ('grouped', 'random'): key = exp + '_fold' ys, gs, ps, packed, s_parts = [], [], {}, [], [] for n in range(5): te = [r for r in rows if int(r['dep'][key]) == n] tr = [r for r in rows if int(r['dep'][key]) != n] train_h, test_h, preds, ln_s, s_const = run(tr, te, f'fold {n + 1}', log, epochs, n) y = [math.log(r['row']['h']) for r in test_h] ys.append(y); gs.append([r['stype'] for r in test_h]) for k, v in preds.items(): ps.setdefault(k, []).append(v) packed += pack(test_h, y, preds) s_parts.append((te, ln_s, s_const)) y = np.concatenate(ys) res['score'] = T.score(y, {k: np.concatenate(v) for k, v in ps.items()}, np.concatenate(gs)) te_all = [r for part in s_parts for r in part[0]] res['S'] = score_s(te_all, np.concatenate([p[1] for p in s_parts]), np.concatenate([np.full(len(p[0]), p[2]) for p in s_parts]), np.array([r['stype'] for r in te_all])) res['compounds_held_out'] = packed else: raise SystemExit('experiments: top, family, grouped, random') log(json.dumps(res['score'], indent=1)) log(json.dumps(res['S'], indent=1)) json.dump(res, open(OUT / f'{exp}.json', 'w'), indent=1) if __name__ == '__main__': main()