"""Row counts and fill rates for the NIMS MDR SuperCon Datasheet, Ver.240322. Inputs (CC BY 4.0, NIMS Materials Database Group), not stored in this repository: 20240322_MDR_OAndM.txt 19,875,381 bytes DOI 10.48505/nims.4487 https://mdr.nims.go.jp/filesets/2347b413-9c15-43b7-90e6-41fe9243b1a5/download sha256 f599ef0040c18521e386f758ee826fe269f67ef369c1a007656035d03ccdfdf6 240322_MDR_Organic.txt 263,372 bytes DOI 10.48505/nims.4487 https://mdr.nims.go.jp/filesets/1a18e447-8ec2-4316-be12-b4acdc885fb8/download sha256 d116848a90d01e48356ed6cf0bb69035fa863a3a559deac1ef188025bdb3b423 primary.tsv 1,962,977 bytes DOI 10.48505/nims.4488 https://mdr.nims.go.jp/filesets/0d651cf6-c52f-4721-a28b-ed661a1dfb4b/download sha256 e2bad520d6008f8c4621c1d0015ee9741bb74477e3941bae2fcf64b1a5747fc5 Usage: python supercon_counts.py /path/to/dir/with/the/three/files > supercon_counts.json Standard library only. Definitions used throughout: filled the cell is not empty after stripping whitespace tc>0 the recommended-Tc column `tc` parses to a number above zero tcn-only `tcn` is filled and none of t1, t2, t3, tcsus, tc is filled composition elements and amounts from the ma1..mj2 and mo1/mo2 columns, normalised to atomic fractions at three decimals; a variable oxygen content such as "7-z" is replaced by its leading number and flagged system the set of element symbols in the row earliest year minimum of `year` over all rows sharing the key """ import collections import hashlib import json import os import re import sys CUTS = [1986, 2008, 2014, 2015, 2019] TC_COLS = ["t1", "t2", "t3", "tcsus", "tc"] SLOTS = "abcdefghij" def filled(v): return v is not None and v.strip() != "" def fnum(v): try: return float(v.strip()) except (ValueError, AttributeError): return None def load_table(path, encoding="utf-8", header_rows=2): """Tab-separated, CRLF row terminator, no quoting. Some cells contain a bare LF.""" with open(path, encoding=encoding, newline="") as f: raw = f.read() sep = "\r\n" if "\r\n" in raw else "\n" lines = raw.split(sep) if lines[-1] == "": lines = lines[:-1] symbols = lines[header_rows - 1].split("\t") rows = [] for line in lines[header_rows:]: cells = line.split("\t") if len(cells) != len(symbols): raise ValueError(f"{path}: row with {len(cells)} cells, expected {len(symbols)}") rows.append(dict(zip(symbols, cells))) return symbols, rows def sha256(path): h = hashlib.sha256() with open(path, "rb") as f: for chunk in iter(lambda: f.read(1 << 20), b""): h.update(chunk) return h.hexdigest() def year_of(r): y = r["year"].strip() return int(y) if re.fullmatch(r"\d{4}", y) else None def journal_year(r): m = re.findall(r"\((\d{4})\)", r["journal"]) return int(m[-1]) if m else None def is_sc(r): return (fnum(r["tc"]) or 0) > 0 def tcn_only(r): return filled(r["tcn"]) and not any(filled(r[c]) for c in TC_COLS) def composition(r): comp = {} variable = False for s in SLOTS: el = r[f"m{s}1"].strip() if not el: continue x = fnum(r[f"m{s}2"]) if x is None: variable = True comp.setdefault(el, None) else: comp[el] = (comp.get(el) or 0) + x if r["mo1"].strip(): amt = r["mo2"].strip() x = fnum(amt) if x is None: m = re.match(r"^(\d+(?:\.\d+)?)", amt) x = float(m.group(1)) if m else None variable = True comp[r["mo1"].strip()] = x return comp, variable def key_comp(comp): num = {e: v for e, v in comp.items() if v is not None and v > 0} tot = sum(num.values()) if tot <= 0: return ("?",) + tuple(sorted(comp)) items = tuple(sorted((e, round(v / tot, 3)) for e, v in num.items())) unknown = tuple(sorted(e for e, v in comp.items() if v is None)) return items + ((("?",) + unknown) if unknown else ()) def key_system(comp): return tuple(sorted(comp)) def families(comp): els = set(comp) out = [] cu_o = "Cu" in els and "O" in els if cu_o: out.append("cuprate") if "Fe" in els and els & {"As", "P", "Se", "Te", "S"} and not cu_o: out.append("iron") if ("Ni" in els and "O" in els and "Cu" not in els and not els & {"Fe", "As", "P", "Se", "S", "Bi", "B", "C", "N", "Te", "Mo", "Sb"}): out.append("nickelate") if els in ({"H", "S"}, {"D", "S"}): out.append("H-S") if els == {"La", "H"}: out.append("La-H") if els == {"Ca", "H"}: out.append("Ca-H") return out def quantile(sorted_vals, p): return sorted_vals[int(round(p * (len(sorted_vals) - 1)))] def main(data_dir): out = {} p_main = os.path.join(data_dir, "20240322_MDR_OAndM.txt") p_org = os.path.join(data_dir, "240322_MDR_Organic.txt") p_prim = os.path.join(data_dir, "primary.tsv") out["sha256"] = {os.path.basename(p): sha256(p) for p in (p_main, p_org, p_prim)} cols, rows = load_table(p_main) n = len(rows) comps = [composition(r) for r in rows] out["oxide_metallic"] = om = {"rows": n, "columns": len(cols)} om["fill"] = {c: sum(filled(r[c]) for r in rows) for c in ["year", "month", "journal", "title", "t1", "t2", "t3", "tcsus", "tcn", "tcwidth", "tc", "tcmeth", "dtcdp", "pmax", "str3", "spaceg", "lata", "shape", "commt"]} om["fill_pct"] = {c: round(100 * v / n, 2) for c, v in om["fill"].items()} om["tc_gt0"] = sum(is_sc(r) for r in rows) om["tc_eq0"] = sum(fnum(r["tc"]) == 0 for r in rows) om["tc_empty"] = sum(not filled(r["tc"]) for r in rows) om["any_tc_col_filled"] = sum(any(filled(r[c]) for c in TC_COLS) for r in rows) om["tcn_only"] = sum(tcn_only(r) for r in rows) om["no_tc_no_tcn"] = sum(not filled(r["tcn"]) and not any(filled(r[c]) for c in TC_COLS) for r in rows) om["tcn_and_some_tc"] = sum(filled(r["tcn"]) and any(filled(r[c]) for c in TC_COLS) for r in rows) om["no_tc_no_tcn_cuprate"] = sum( 1 for r, (c, _) in zip(rows, comps) if not filled(r["tcn"]) and not any(filled(r[k]) for k in TC_COLS) and "Cu" in c and "O" in c) om["tcmeth_values"] = dict(collections.Counter(r["tcmeth"].strip() for r in rows if filled(r["tcmeth"]))) om["pmax_rows"] = [[r["num"], r["element"], r["year"], r["tc"], r["pmax"]] for r in rows if filled(r["pmax"])] years = [year_of(r) for r in rows] om["year"] = { "filled": sum(y is not None for y in years), "min": min(y for y in years if y), "max": max(y for y in years if y), "by_year_2015_on": {str(y): sum(1 for v in years if v == y) for y in range(2015, 2024)}, "by_decade": {str(d): sum(1 for v in years if v is not None and v // 10 * 10 == d) for d in range(1910, 2030, 10)}, "differs_from_journal_string": sum( 1 for r, y in zip(rows, years) if y and journal_year(r) and y != journal_year(r)), "differs_by_2_or_more": sum( 1 for r, y in zip(rows, years) if y and journal_year(r) and abs(y - journal_year(r)) >= 2), "crosses_cut": {str(c): sum(1 for r, y in zip(rows, years) if y and journal_year(r) and (y < c) != (journal_year(r) < c)) for c in CUTS}, } tcn = sorted(fnum(r["tcn"]) for r in rows if filled(r["tcn"])) floor = sorted(fnum(r["tcn"]) for r in rows if tcn_only(r)) om["tcn"] = { "n": len(tcn), "zeros": sum(v == 0 for v in tcn), "quantiles_K": {str(p): quantile(tcn, p) for p in (0.05, 0.1, 0.25, 0.5, 0.75, 0.9, 0.95, 0.99)}, "max_K": tcn[-1], "bins": {"0": sum(v == 0 for v in tcn), "(0,0.1]": sum(0 < v <= 0.1 for v in tcn), "(0.1,0.5]": sum(0.1 < v <= 0.5 for v in tcn), "(0.5,1]": sum(0.5 < v <= 1 for v in tcn), "(1,2]": sum(1 < v <= 2 for v in tcn), "(2,4.2]": sum(2 < v <= 4.2 for v in tcn), "(4.2,10]": sum(4.2 < v <= 10 for v in tcn), "(10,20]": sum(10 < v <= 20 for v in tcn), "(20,77]": sum(20 < v <= 77 for v in tcn), ">77": sum(v > 77 for v in tcn)}, "most_common": collections.Counter(tcn).most_common(8), "tcn_only": {"n": len(floor), "zeros": sum(v == 0 for v in floor), "median_K": quantile(floor, 0.5), "mean_K": round(sum(floor) / len(floor), 3), "le_1K": sum(0 < v <= 1 for v in floor), "le_2K": sum(0 < v <= 2 for v in floor), "le_4.2K": sum(0 < v <= 4.2 for v in floor), "gt_4.2K": sum(v > 4.2 for v in floor), "gt_10K": sum(v > 10 for v in floor), "ge_77K": sum(v >= 77 for v in floor)}, "tcn_eq0_with_tc_gt0": sum(fnum(r["tcn"]) == 0 and is_sc(r) for r in rows), "tcn_eq0_with_P_induced_comment": sum( fnum(r["tcn"]) == 0 and "p-induced" in r["commt"].lower() for r in rows), "tcn_only_by_decade": {str(d): sum(1 for r, y in zip(rows, years) if tcn_only(r) and y is not None and y // 10 * 10 == d) for d in range(1920, 2030, 10)}, "tcn_only_undated": sum(1 for r, y in zip(rows, years) if tcn_only(r) and y is None), } pat = re.compile(r"GPa|kbar|P-induced|pressure|under P\b|high-P|Mbar", re.I) om["pressure_text"] = { "commt_rows": sum(bool(pat.search(r["commt"])) for r in rows), "commt_rows_tc_gt0": sum(bool(pat.search(r["commt"])) and is_sc(r) for r in rows), "P_induced_rows": sum("p-induced" in r["commt"].lower() for r in rows), "P_induced_rows_tc_gt0": sum("p-induced" in r["commt"].lower() and is_sc(r) for r in rows), } om["tc_tail"] = {f">={t}K": sum((fnum(r["tc"]) or 0) >= t for r in rows) for t in (135, 150, 200, 250, 273)} om["rows_any_tc_col_ge_150K"] = [ [r["num"], r["element"], r["name"], r["year"], r["tc"], r["journal"]] for r in rows if max((fnum(r[c]) or 0) for c in TC_COLS) >= 150] def distinct(keyfn): first, sc, cnt = {}, {}, collections.Counter() for r, (c, _), y in zip(rows, comps, years): k = keyfn(r, c) cnt[k] += 1 if y is not None: first[k] = min(first.get(k, 9999), y) sc[k] = sc.get(k, False) or is_sc(r) d = {"distinct": len(cnt), "dated": len(first), "undated": len(cnt) - len(first), "with_tc_gt0": sum(sc.values()), "cuts": {}} for cut in CUTS: d["cuts"][str(cut)] = { "earliest_before": sum(1 for k in first if first[k] < cut), "earliest_before_tc_gt0": sum(1 for k in first if first[k] < cut and sc[k]), "earliest_on_or_after": sum(1 for k in first if first[k] >= cut), "earliest_on_or_after_tc_gt0": sum(1 for k in first if first[k] >= cut and sc[k]), } return d, first om["distinct_element_string"], _ = distinct(lambda r, c: r["element"].strip()) om["distinct_composition"], first_comp = distinct(lambda r, c: key_comp(c)) om["distinct_system"], first_sys = distinct(lambda r, c: key_system(c)) om["rows_fully_numeric_composition"] = sum(not v for _, v in comps) om["distinct_composition_fully_numeric"] = len({key_comp(c) for c, v in comps if not v}) om["rows_by_cut"] = {} for cut in CUTS: test = [(r, c) for r, (c, _), y in zip(rows, comps, years) if y is not None and y >= cut] test_sc = [(r, c) for r, c in test if is_sc(r)] train_sc = [fnum(r["tc"]) for r, y in zip(rows, years) if y is not None and y < cut and is_sc(r)] om["rows_by_cut"][str(cut)] = { "before": sum(1 for y in years if y is not None and y < cut), "before_tc_gt0": len(train_sc), "before_tcn_only": sum(1 for r, y in zip(rows, years) if y is not None and y < cut and tcn_only(r)), "on_or_after": len(test), "on_or_after_tc_gt0": len(test_sc), "on_or_after_tcn_only": sum(tcn_only(r) for r, _ in test), "test_tc_gt0_rows_in_system_first_seen_on_or_after_cut": sum( 1 for r, c in test_sc if first_sys[key_system(c)] >= cut), "test_tc_gt0_rows_with_composition_first_seen_on_or_after_cut": sum( 1 for r, c in test_sc if first_comp[key_comp(c)] >= cut), "train_top5_tc": sorted(train_sc, reverse=True)[:5], "test_tc_ge": {str(t): sum(fnum(r["tc"]) >= t for r, _ in test_sc) for t in (23, 40, 77, 100, 135, 200)}, "train_tc_ge": {str(t): sum(v >= t for v in train_sc) for t in (23, 40, 77, 100, 135, 200)}, } fam = collections.defaultdict(lambda: {"rows": 0, "tc_gt0": 0, "tcn_only": 0, "cuts": {}}) fam_first = collections.defaultdict(dict) fam_sc = collections.defaultdict(dict) for r, (c, _), y in zip(rows, comps, years): for f in families(c): k = key_comp(c) fam[f]["rows"] += 1 fam[f]["tc_gt0"] += is_sc(r) fam[f]["tcn_only"] += tcn_only(r) if y is not None: fam_first[f][k] = min(fam_first[f].get(k, 9999), y) fam_sc[f][k] = fam_sc[f].get(k, False) or is_sc(r) for cut in CUTS: d = fam[f]["cuts"].setdefault(str(cut), {"rows": 0, "tc_gt0": 0, "tcn_only": 0, "max_tc": 0}) if y is not None and y >= cut: d["rows"] += 1 d["tc_gt0"] += is_sc(r) d["tcn_only"] += tcn_only(r) d["max_tc"] = max(d["max_tc"], fnum(r["tc"]) or 0) for f in ("cuprate", "iron", "nickelate", "H-S", "La-H", "Ca-H"): fam[f]["distinct_compositions"] = len(fam_sc[f]) for cut in CUTS: d = fam[f]["cuts"].setdefault(str(cut), {"rows": 0, "tc_gt0": 0, "tcn_only": 0, "max_tc": 0}) d["compositions_first_seen"] = sum(1 for k, y in fam_first[f].items() if y >= cut) d["compositions_first_seen_tc_gt0"] = sum( 1 for k, y in fam_first[f].items() if y >= cut and fam_sc[f][k]) om["families"] = {f: fam[f] for f in ("cuprate", "iron", "nickelate", "H-S", "La-H", "Ca-H")} om["hydrogen_rich_rows"] = [ [r["num"], r["element"], r["name"], r["year"], r["tc"], r["tcn"], r["pmax"]] for r, (c, _) in zip(rows, comps) if (lambda num: sum(num.values()) > 0 and (num.get("H", 0) + num.get("D", 0)) / sum(num.values()) >= 0.5)( {e: v for e, v in c.items() if v is not None and v > 0}) and (year_of(r) or 0) >= 2000] conflict_keys = {key_comp(c) for r, (c, _) in zip(rows, comps) if is_sc(r)} neg_keys = collections.Counter(key_comp(c) for r, (c, _) in zip(rows, comps) if tcn_only(r)) om["label_conflicts"] = { "compositions_with_tcn_only_row": len(neg_keys), "of_which_also_tc_gt0": sum(1 for k in neg_keys if k in conflict_keys), "tcn_only_rows_in_conflict": sum(v for k, v in neg_keys.items() if k in conflict_keys), } ocols, orows = load_table(p_org) orows_real = [r for r in orows if filled(r["num"])] oy = [int(r["year"]) for r in orows_real if re.fullmatch(r"\d{4}", r["year"].strip())] out["organic"] = { "rows_in_file": len(orows), "rows_with_num": len(orows_real), "columns": len(ocols), "fill": {c: sum(filled(r[c]) for r in orows) for c in ["year", "journal", "tc", "tcmax", "pmax", "pcrit", "tcn", "tcmeth", "dtcdp"]}, "year_min": min(oy), "year_max": max(oy), "pcrit_eq0": sum(fnum(r["pcrit"]) == 0 for r in orows), "pcrit_gt0": sum((fnum(r["pcrit"]) or 0) > 0 for r in orows), "tc_gt0": sum((fnum(r["tc"]) or 0) > 0 for r in orows), "tcn_only": sum(filled(r["tcn"]) and not filled(r["tc"]) and not filled(r["tcmax"]) for r in orows), "year_cuts": {str(c): {"before": sum(y < c for y in oy), "on_or_after": sum(y >= c for y in oy)} for c in CUTS}, } with open(p_prim, encoding="utf-8", newline="") as f: plines = f.read().split("\n") if plines[-1] == "": plines = plines[:-1] phead = plines[2].split("\t") prow = [dict(zip(phead, line.split("\t"))) for line in plines[3:]] main_tc = {r["num"]: r["tc"] for r in rows if filled(r["tc"])} out["primary_tsv"] = { "lines": len(plines), "header_rows": 3, "data_rows": len(prow), "columns": phead, "tc_gt0": sum((fnum(p["tc"]) or 0) > 0 for p in prow), "tc_eq0": sum(fnum(p["tc"]) == 0 for p in prow), "same_num_set_as_main_rows_with_tc_filled": {p["num"] for p in prow} == set(main_tc), "tc_mismatches_vs_main": sum(fnum(p["tc"]) != fnum(main_tc.get(p["num"], "")) for p in prow), "max_tc": max(fnum(p["tc"]) for p in prow), } json.dump(out, sys.stdout, indent=1, sort_keys=False) sys.stdout.write("\n") if __name__ == "__main__": main(sys.argv[1] if len(sys.argv) > 1 else os.environ.get("SUPERCON_DIR", "."))