"""How much the splits of the first deposit leaked, and what its scores are without the compounds affected. A held-out hydride counts as exposed if a labelled hydride of the same structure type (labelled independently of the origin, model/structure_types.py) was on the training side, and as duplicated if another record of the same compound was. Output: model/results/leak_check.json Run: .venv/bin/python model/leak_check.py """ import csv import json import math import pathlib import numpy as np ROOT = pathlib.Path(__file__).resolve().parents[1] def main(): dep = {r['id']: r for r in csv.DictReader(open(ROOT / 'model' / 'deposit' / 'splits.csv'))} st = {r['id']: r for r in csv.DictReader(open(ROOT / 'model' / 'data' / 'structure_types.csv'))} lab = {r['id']: r for r in csv.DictReader(open(ROOT / 'model' / 'data' / 'labels.csv'))} out = {} def sides(col, test_value, by_fold=None): if by_fold is None: te = [i for i, r in dep.items() if r[col] == test_value] tr = [i for i, r in dep.items() if r[col] != test_value] return [(te, tr)] return [([i for i, r in dep.items() if r[col] == str(k)], [i for i, r in dep.items() if r[col] != str(k)]) for k in range(5)] for name, parts in (('top', sides('top', 'test')), ('a2mh6', sides('a2mh6', 'test')), ('grouped', sides('grouped_fold', None, True)), ('random', sides('random_fold', None, True))): n = exposed = duplicated = 0 flagged = {} for te, tr in parts: tr_h = [i for i in tr if dep[i]['has_h'] == '1'] types = {st[i]['structure_type'] for i in tr_h} comps = {st[i]['compound'] for i in tr} for i in te: if dep[i]['has_h'] != '1': continue n += 1 e, d = st[i]['structure_type'] in types, st[i]['compound'] in comps exposed += e duplicated += d flagged[i] = (e, d) out[name] = dict(held_out_hydrides=n, with_their_structure_type_in_training=exposed, with_another_record_of_the_compound_in_training=duplicated) res = ROOT / 'model' / 'results' / f'{name}.json' if res.exists(): r = json.load(open(res)) comp = r.get('compounds_held_out') if comp and 'id' not in comp[0]: # the first results stored formula and h only: recover identifiers by matching both pool = {} for i in flagged: pool.setdefault(lab[i]['formula'], []).append(i) for c in comp: cand = pool.get(c['formula'], []) best = min(cand, key=lambda i: abs(float(lab[i]['h']) / c['h'] - 1), default=None) c['id'] = best if best is not None and abs(float(lab[best]['h']) / c['h'] - 1) < 1e-4 else None if c['id']: cand.remove(best) if comp: def mae(rows, key): return float(np.mean([abs(math.log(c[key] / c['h'])) for c in rows])) if rows else None for label, keep in (('all', lambda c: True), ('not_exposed', lambda c: c['id'] and not flagged[c['id']][0]), ('not_duplicated', lambda c: c['id'] and not flagged[c['id']][1]), ('exposed_only', lambda c: c['id'] and flagged[c['id']][0])): rows = [c for c in comp if keep(c)] if rows: best = min(mae(rows, k) for k in ('gas', 'dos', 'const')) out[name][label] = dict(n=len(rows), trees=mae(rows, 'trees'), graph=mae(rows, 'graph'), best_baseline=best, trees_ratio=mae(rows, 'trees') / best, graph_ratio=mae(rows, 'graph') / best) json.dump(out, open(ROOT / 'model' / 'results' / 'leak_check.json', 'w'), indent=1) print(json.dumps(out, indent=1)) if __name__ == '__main__': main()