kozo2 commited on
Commit
f574cd4
·
verified ·
1 Parent(s): a037de7

Upload config.py with huggingface_hub

Browse files
Files changed (1) hide show
  1. config.py +101 -0
config.py ADDED
@@ -0,0 +1,101 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Per-source-table paths and node metadata, so one set of scripts serves them all.
2
+
3
+ Every script takes the source table as its first argument (default edge_ML) and
4
+ resolves its own files from it:
5
+
6
+ graph/<src>_filtered_expected_ge5_pyg.pt build_graph.py
7
+ <src>_filtered_expected_ge5_n2v.pt node2vec_model.py
8
+ figures/umap_coords_<src>_filtered.npy umap_species.py (cache)
9
+ figures/umap_species_<src>_filtered_<mode>.png
10
+
11
+ Node metadata (species) lives in graph.duckdb, split across two tables by id
12
+ namespace: MetaboLights ids (MTBLS...) in nodes_ML, Metabolomics Workbench ids
13
+ (ST...) in nodes_MW. edge_MLvsMW spans both, so the lookup is their union.
14
+ """
15
+
16
+ import argparse
17
+ import os
18
+ import pathlib
19
+
20
+ SOURCE_TABLES = ("edge_ML", "edge_MLvsMW", "edge_MW_1", "edge_MW_2", "edge_MW_3")
21
+
22
+ DATA_DIR = pathlib.Path(os.environ.get(
23
+ "DATA_DIR", str(pathlib.Path(__file__).resolve().parent.parent)))
24
+ DB_PATH = str(DATA_DIR / "edges_filtered.duckdb")
25
+ NODE_DB_PATH = str(DATA_DIR / "graph.duckdb") # nodes_ML / nodes_MW live here
26
+ NODES_PARQUET = DATA_DIR / "data" / "nodes_expected_ge5.parquet"
27
+ EDGE_TABLE = "edges_expected_ge5"
28
+ GRAPH_DIR = DATA_DIR / "graph"
29
+ FIGURES_DIR = DATA_DIR / "figures"
30
+
31
+
32
+ class Paths:
33
+ """The files belonging to one source table."""
34
+
35
+ def __init__(self, source_table: str):
36
+ if source_table not in SOURCE_TABLES:
37
+ raise SystemExit(f"unknown source table {source_table!r}; "
38
+ f"expected one of {', '.join(SOURCE_TABLES)}")
39
+ self.source_table = source_table
40
+ GRAPH_DIR.mkdir(parents=True, exist_ok=True)
41
+ FIGURES_DIR.mkdir(parents=True, exist_ok=True)
42
+ stem = f"{source_table}_filtered_expected_ge5"
43
+ self.graph = str(GRAPH_DIR / f"{stem}_pyg.pt")
44
+ self.embeddings = str(DATA_DIR / f"{stem}_n2v.pt")
45
+ self.umap_npy = str(FIGURES_DIR / f"umap_coords_{source_table}_filtered.npy")
46
+
47
+ def figure(self, mode: str, color_by: str = "species") -> str:
48
+ return str(FIGURES_DIR /
49
+ f"umap_{color_by}_{self.source_table}_filtered_{mode}.png")
50
+
51
+
52
+ def add_source_arg(ap: argparse.ArgumentParser) -> None:
53
+ ap.add_argument("--source-table", default="edge_ML", choices=SOURCE_TABLES,
54
+ help="which source_table of edges_expected_ge5 to use")
55
+
56
+
57
+ # id namespace -> the database it comes from. MetaboLights study ids start with
58
+ # MTBLS, Metabolomics Workbench ids with ST; edge_MLvsMW joins one of each.
59
+ DATABASES = (("MTBLS", "MetaboLights (MTBLS)"), ("ST", "Metabolomics Workbench (ST)"))
60
+
61
+
62
+ def database_for(node_ids: list) -> list:
63
+ """Source database per node id, in node_ids order."""
64
+ out = []
65
+ for i in node_ids:
66
+ for prefix, name in DATABASES:
67
+ if i.startswith(prefix):
68
+ out.append(name)
69
+ break
70
+ else:
71
+ raise SystemExit(f"id {i!r} matches no known database prefix")
72
+ return out
73
+
74
+
75
+ def node_property_query(con, column: str = "species") -> str:
76
+ """SQL returning (id, <column>) for every node, from whichever source exists.
77
+
78
+ The released dataset ships data/nodes_expected_ge5.parquet, which holds the
79
+ property rows for exactly the nodes in edges_expected_ge5 — enough for every
80
+ graph built from that table, and the only source a downloader has. The
81
+ original working tree instead has graph.duckdb, where the rows are split
82
+ across nodes_ML (MTBLS ids) and nodes_MW (ST ids).
83
+ """
84
+ if NODES_PARQUET.exists():
85
+ return f"SELECT id, {column} FROM '{NODES_PARQUET}'"
86
+ if pathlib.Path(NODE_DB_PATH).exists():
87
+ con.execute(f"ATTACH IF NOT EXISTS '{NODE_DB_PATH}' AS g (READ_ONLY)")
88
+ return (f"SELECT id, {column} FROM g.nodes_ML "
89
+ f"UNION ALL SELECT id, {column} FROM g.nodes_MW")
90
+ raise SystemExit(f"no node properties found: expected {NODES_PARQUET} "
91
+ f"or {NODE_DB_PATH}")
92
+
93
+
94
+ def species_for(con, node_ids: list) -> list:
95
+ """species per node id, in node_ids order. Raises if any id is unknown."""
96
+ by_id = dict(con.execute(node_property_query(con, "species")).fetchall())
97
+ missing = [i for i in node_ids if i not in by_id]
98
+ if missing:
99
+ raise SystemExit(f"{len(missing):,} nodes have no metadata row, "
100
+ f"e.g. {missing[:3]}")
101
+ return [by_id[i] for i in node_ids]