From 89fe0505b3631f8c10253934bc8fd005900ba90b Mon Sep 17 00:00:00 2001 From: Jeremy Date: Mon, 21 Sep 2026 11:50:01 -0700 Subject: [PATCH] Add the cNF organoid dataset: samples and omics --- coderbuild/cnf/01-samples-cnf.py | 372 +++++++++++++++++++++++++++++ coderbuild/cnf/02-omics-cnf.py | 389 +++++++++++++++++++++++++++++++ coderbuild/cnf/README.md | 169 ++++++++++++++ coderbuild/cnf/build_omics.sh | 20 ++ coderbuild/cnf/build_samples.sh | 15 ++ coderbuild/cnf/cnf_utils.py | 117 ++++++++++ coderbuild/cnf/requirements.txt | 9 + coderbuild/docker/Dockerfile.cnf | 74 ++++++ 8 files changed, 1165 insertions(+) create mode 100644 coderbuild/cnf/01-samples-cnf.py create mode 100644 coderbuild/cnf/02-omics-cnf.py create mode 100644 coderbuild/cnf/README.md create mode 100644 coderbuild/cnf/build_omics.sh create mode 100644 coderbuild/cnf/build_samples.sh create mode 100644 coderbuild/cnf/cnf_utils.py create mode 100644 coderbuild/cnf/requirements.txt create mode 100644 coderbuild/docker/Dockerfile.cnf diff --git a/coderbuild/cnf/01-samples-cnf.py b/coderbuild/cnf/01-samples-cnf.py new file mode 100644 index 00000000..41bb14ea --- /dev/null +++ b/coderbuild/cnf/01-samples-cnf.py @@ -0,0 +1,372 @@ +#!/usr/bin/env python3 +""" +01-samples-cnf.py: generate cnf_samples.csv for the cNF organoid drug screen. + +Specimens are gathered from four sources: + + 1. Drug-screen file index on Synapse (syn51301431, dataType='drug screen') + 2. Global proteomics matrix (env CNF_GLOBAL_PROT_SYN_ID, default syn74815895) + 3. RNA discovery file (env CNF_RNA_DISCOVERY_SYN_ID, default syn71333780). + This is the corrected-abundance file containing every RNA condition + processed for the project (organoid, tumor tissue, drug-treated + organoid, skin), used here for sample discovery only — NOT to be + confused with CNF_RNA_TPM_SYN_ID, which build_omics.py reads for the + actual TPM data and may be filtered to organoids only. + 4. Normal Skin Synapse folder (env CNF_NORMAL_SKIN_SYN_ID, default syn74284682) + +Each specimen is classified as one of: + - "organoid" → model_type = "patient derived organoid" + - "treated_organoid" → model_type = "patient derived organoid" + (drug-treated organoid; canonical ID preserves the + treatment label, e.g. NF0021_T1_Onalespid_1uM) + - "tumor_tissue" → model_type = "tumor" + - "normal_skin" → model_type = "normal_tissue" + +Tissue and organoid samples for the same tumor are kept as separate rows with +distinct improve_sample_ids — they have different biology (fresh tumor vs. +cultured organoid) and the schema model_types differ. Treated and untreated +organoids of the same tumor are also separate rows because their canonical +IDs differ (treated samples preserve the treatment label). + +Cohort assignment is intentionally omitted. The upstream omics data is +batch-corrected and per-modality cohort membership disagrees for some +specimens, so a single per-sample cohort label cannot be assigned cleanly. +The `other_id_source` field carries a uniform project label. + +Sample IDs continue from the previous samples file: max(improve_sample_id) + 1. + +Usage: + python 01-samples-cnf.py [--output /tmp/cnf_samples.csv] +""" + +import argparse +import logging +import os +import re +import sys + +import pandas as pd +import synapseclient + +from cnf_utils import ( + MODEL_TYPE_MAP, + classify_specimen, + patient_from_specimen, +) + + +# ---------------------------------------------------------------------------- +# Constants +# ---------------------------------------------------------------------------- +DRUG_SCREEN_TABLE = "syn51301431" +GLOBAL_PROT_SYN_ID = os.environ.get("CNF_GLOBAL_PROT_SYN_ID", "syn74815895") +# RNA file used for SPECIMEN DISCOVERY only — needs to contain all +# conditions (organoid + tissue + treated + skin). The corrected-abundance +# file at syn71333780 fits this. Also used by build_omics.py as the TPM +# data source (the column is named correctedAbundance but stores TPM). +RNA_DISCOVERY_SYN_ID = os.environ.get("CNF_RNA_DISCOVERY_SYN_ID", "syn71333780") +NORMAL_SKIN_SYN_ID = os.environ.get("CNF_NORMAL_SKIN_SYN_ID", "syn74284682") + +DEFAULT_OUTPUT = "/tmp/cnf_samples.csv" + +CANCER_TYPE = "Cutaneous Neurofibroma" +SPECIES = "Homo sapiens (Human)" +OTHER_ID_SOURCE = "NF1_cNF_Project" + + +# ---------------------------------------------------------------------------- +# Helpers +# ---------------------------------------------------------------------------- +def configure_logging(): + logging.basicConfig( + format="[%(asctime)s] [%(levelname)s] %(message)s", + datefmt="%Y-%m-%d %H:%M:%S", + level=logging.INFO, + ) + + +def get_starting_id(prev_samples_path: str) -> int: + """Read previous samples file and return max(improve_sample_id) + 1.""" + if not os.path.exists(prev_samples_path): + logging.warning("Previous samples file %s not found; starting at 1.", + prev_samples_path) + return 1 + + df = pd.read_csv(prev_samples_path) + if "improve_sample_id" not in df.columns or df.empty: + logging.warning("No improve_sample_id column or empty file; starting at 1.") + return 1 + + max_id = int(pd.to_numeric(df["improve_sample_id"], errors="coerce").max()) + logging.info("Previous samples max improve_sample_id = %d", max_id) + return max_id + 1 + + +def merge_specimen_dicts(*dicts: dict[str, str]) -> dict[str, str]: + """Union per-source dicts; warn on type disagreements (first-seen wins).""" + merged: dict[str, str] = {} + for d in dicts: + for spec, stype in d.items(): + if spec in merged and merged[spec] != stype: + logging.warning( + "Type conflict for %s: %s (kept) vs %s (ignored)", + spec, merged[spec], stype, + ) + continue + merged.setdefault(spec, stype) + return merged + + +# ---------------------------------------------------------------------------- +# Source-specific specimen extraction +# ---------------------------------------------------------------------------- +def specimens_from_drug_screen(syn) -> dict[str, str]: + """Pull canonical_id → sample_type for drug-screen specimens.""" + query = ( + "select id, individualID, specimenID " + f"from {DRUG_SCREEN_TABLE} " + "where dataType='drug screen'" + ) + logging.info("Querying %s for drug-screen specimens", DRUG_SCREEN_TABLE) + table = syn.tableQuery(query).asDataFrame() + raw = table["specimenID"].dropna().astype(str).unique().tolist() + out: dict[str, str] = {} + for r in raw: + result = classify_specimen(r) + if result: + cid, stype = result + out[cid] = stype + by_type = _count_by_type(out) + logging.info(" Drug screen: %s", by_type) + return out + + +def _count_by_type(d: dict[str, str]) -> dict[str, int]: + out: dict[str, int] = {} + for stype in d.values(): + out[stype] = out.get(stype, 0) + 1 + return out + + +def _detect_specimen_column(df: pd.DataFrame) -> str: + """Return the name of the specimen column in an omics file.""" + candidates = ["Specimen", "specimen", "specimen_norm", "sample_id", + "specimenID", "Sample"] + for c in candidates: + if c in df.columns: + return c + raise KeyError( + f"No specimen column among {candidates} in columns {list(df.columns)[:15]}" + ) + + +def _read_synapse_table(syn, syn_id: str) -> pd.DataFrame: + """Fetch a Synapse file and read it as CSV/TSV.""" + logging.info(" Fetching %s", syn_id) + entity = syn.get(syn_id) + path = entity.path + if path.endswith(".tsv") or path.endswith(".tsv.gz"): + return pd.read_csv(path, sep="\t") + return pd.read_csv(path) + + +def specimens_from_proteomics(syn) -> dict[str, str]: + """Pull canonical_id → sample_type from the global proteomics matrix.""" + if not GLOBAL_PROT_SYN_ID: + logging.warning("CNF_GLOBAL_PROT_SYN_ID not set; skipping proteomics") + return {} + + logging.info("Reading proteomics specimens") + try: + df = _read_synapse_table(syn, GLOBAL_PROT_SYN_ID) + col = _detect_specimen_column(df) + raw = df[col].dropna().astype(str).unique().tolist() + out: dict[str, str] = {} + for r in raw: + result = classify_specimen(r) + if result: + cid, stype = result + out[cid] = stype + logging.info(" Proteomics: %s", _count_by_type(out)) + return out + except Exception as exc: + logging.warning(" Proteomics read failed: %s", exc) + return {} + + +def specimens_from_rna_discovery(syn) -> dict[str, str]: + """Discover RNA specimens from a comprehensive long-format RNA file. + + Reads CNF_RNA_DISCOVERY_SYN_ID (default syn71333780) — the corrected- + abundance RNA file that contains every condition processed for the + project: organoid + tumor tissue + drug-treated organoid + skin. + + This is intentionally a different file from the one read by + build_omics.py (CNF_RNA_TPM_SYN_ID, which is the gene-level TPM matrix + and may be filtered to organoids only). Sample discovery and data + ingestion have different requirements: discovery wants the maximum + sample list; ingestion wants TPM values. + + Specimens are passed through classify_specimen, so drug-treated + samples like NF0021_T1_Onalespid.1uM get caught regardless of their + condition_norm value. + """ + if not RNA_DISCOVERY_SYN_ID: + logging.warning("CNF_RNA_DISCOVERY_SYN_ID not set; skipping RNA discovery") + return {} + + logging.info("Reading RNA discovery file (%s)", RNA_DISCOVERY_SYN_ID) + try: + df = _read_synapse_table(syn, RNA_DISCOVERY_SYN_ID) + col = _detect_specimen_column(df) + + raw = df[col].dropna().astype(str).unique().tolist() + out: dict[str, str] = {} + for r in raw: + result = classify_specimen(r) + if result: + cid, stype = result + out[cid] = stype + logging.info(" RNA discovery: %s", _count_by_type(out)) + return out + except Exception as exc: + logging.warning(" RNA discovery read failed: %s", exc) + return {} + + +def specimens_from_normal_skin_tree(syn) -> dict[str, str]: + """Discover skin specimens by listing the Normal Skin Synapse folder. + + The "Normal Skin" tree contains one immediate child per patient (folders + named NF0017, NF0018, ...) plus, in some cases, top-level skin FASTQ + files like NF0035-SKIN-1-04-16-2025_*.fastq.gz. Both forms encode the + patient ID at the start, so we match `^NF\\d{4}` against each child name + and emit one canonical NFxxxx_skin entry per unique patient. + + This is intentionally one API call deep — it avoids walking the whole + WGS tree (which has hundreds of empty lane-stub folders). For most + cases this captures every patient with a matched normal. + """ + if not NORMAL_SKIN_SYN_ID: + logging.warning("CNF_NORMAL_SKIN_SYN_ID not set; skipping Normal Skin tree") + return {} + + logging.info("Listing Normal Skin tree (%s)", NORMAL_SKIN_SYN_ID) + out: dict[str, str] = {} + try: + for child in syn.getChildren(NORMAL_SKIN_SYN_ID): + name = str(child.get("name", "")) + m = re.match(r"^(NF\d{4})", name) + if m: + patient = m.group(1) + out[f"{patient}_skin"] = "normal_skin" + logging.info(" Normal Skin tree: %s", _count_by_type(out)) + except Exception as exc: + logging.warning(" Normal Skin tree listing failed: %s", exc) + return out + + +# ---------------------------------------------------------------------------- +# Sample table construction +# ---------------------------------------------------------------------------- +def build_samples_table(specimens: dict[str, str], start_id: int) -> pd.DataFrame: + """Build the schema-compliant Sample table from {canonical_id: sample_type}. + + Sort order: organoid → treated_organoid → tumor_tissue → normal_skin + (each by ID). Drug-modelable untreated organoids get the lowest + improve_sample_ids — useful for downstream filtering. + """ + sort_priority = { + "organoid": 0, + "treated_organoid": 1, + "tumor_tissue": 2, + "normal_skin": 3, + } + sorted_specimens = sorted( + specimens.items(), + key=lambda kv: (sort_priority.get(kv[1], 99), kv[0]), + ) + + rows = [] + next_id = start_id + for specimen, stype in sorted_specimens: + patient = patient_from_specimen(specimen) + rows.append({ + "improve_sample_id": next_id, + "other_id": specimen, + "other_id_source": OTHER_ID_SOURCE, + "common_name": specimen, + "cancer_type": CANCER_TYPE, + "other_names": patient, + "species": SPECIES, + "model_type": MODEL_TYPE_MAP[stype], + }) + next_id += 1 + + df = pd.DataFrame(rows) + logging.info("Built %d sample rows; ID range %d–%d", + len(df), start_id, next_id - 1) + return df + + +# ---------------------------------------------------------------------------- +# Main +# ---------------------------------------------------------------------------- +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument( + "prev_samples", + help="Path to previous samples CSV (used to find max improve_sample_id)", + ) + parser.add_argument("--output", default=DEFAULT_OUTPUT, + help=f"Output CSV path (default: {DEFAULT_OUTPUT})") + args = parser.parse_args() + + configure_logging() + + syn = synapseclient.Synapse() + syn.login() + + # Pull specimens from each source as canonical_id → sample_type dicts + drug_d = specimens_from_drug_screen(syn) + prot_d = specimens_from_proteomics(syn) + rna_d = specimens_from_rna_discovery(syn) + skin_d = specimens_from_normal_skin_tree(syn) + + all_specimens = merge_specimen_dicts(drug_d, prot_d, rna_d, skin_d) + logging.info( + "Union: %d total specimens — %s", + len(all_specimens), _count_by_type(all_specimens), + ) + + # Per-source uniqueness diagnostics + drug_set, prot_set, rna_set, skin_set = ( + set(drug_d), set(prot_d), set(rna_d), set(skin_d), + ) + only_drug = drug_set - prot_set - rna_set - skin_set + only_prot = prot_set - drug_set - rna_set - skin_set + only_rna = rna_set - drug_set - prot_set - skin_set + only_skin = skin_set - drug_set - prot_set - rna_set + if only_drug: + logging.info(" Only in drug screen: %s", sorted(only_drug)) + if only_prot: + logging.info(" Only in proteomics: %s", sorted(only_prot)) + if only_rna: + logging.info(" Only in RNA: %s", sorted(only_rna)) + if only_skin: + logging.info(" Only in Normal Skin tree: %s", sorted(only_skin)) + + if not all_specimens: + logging.error("No specimens found from any source; aborting") + sys.exit(1) + + start_id = get_starting_id(args.prev_samples) + samples = build_samples_table(all_specimens, start_id) + + os.makedirs(os.path.dirname(args.output) or ".", exist_ok=True) + samples.to_csv(args.output, index=False) + logging.info("Wrote %s (%d rows)", args.output, len(samples)) + + +if __name__ == "__main__": + sys.exit(main() or 0) diff --git a/coderbuild/cnf/02-omics-cnf.py b/coderbuild/cnf/02-omics-cnf.py new file mode 100644 index 00000000..bdbe0fac --- /dev/null +++ b/coderbuild/cnf/02-omics-cnf.py @@ -0,0 +1,389 @@ +#!/usr/bin/env python3 +""" +02-omics-cnf.py: generate cnf_transcriptomics.csv, cnf_proteomics.csv, + and cnf_phosphoproteomics.csv. + +For the cNF dataset: + +* Transcriptomics: raw gene-level TPM from per-cohort salmon.merged.gene_tpm.tsv + files (wide matrix: rows = genes, columns = specimens). Default cohorts: + syn66352931 — cNF_Cohort_1 (NF0017–NF0023) + syn70765053 — cNF_Cohort_2 (NF0025, NF0027, NF0035) + Override via env var CNF_RNA_COHORT_SYN_IDS (comma-separated Synapse IDs). + +* Proteomics: batch-corrected global proteomics from syn74815895 + (correctedAbundance column). Override via env var CNF_GLOBAL_PROT_SYN_ID. + +* Phosphoproteomics: batch-corrected phospho-proteomics from syn70078415 + (correctedAbundance column). Mapped to phosphosite_ids via a phosphosites.csv + reference file. Override via env var CNF_PHOSPHO_SYN_ID. + +* Protocol optimization samples are excluded from RNA via DROP_SAMPLE_SUBSTRINGS. + +Specimens are canonicalized via cnf_utils so that values like +"NF0018.T1.organoid" resolve to the canonical IDs in cnf_samples.csv. +Source rows whose specimen doesn't match any entry in cnf_samples.csv are +dropped silently via the inner join to sample_map. + +Usage: + python 02-omics-cnf.py + [--out_rna /tmp/cnf_transcriptomics.csv] + [--out_prot /tmp/cnf_proteomics.csv] + [--out_phospho /tmp/cnf_phosphoproteomics.csv] +""" + +import argparse +import logging +import os +import re +import sys + +import pandas as pd +import synapseclient + +from cnf_utils import canonicalize_specimen_column + + +# ---------------------------------------------------------------------------- +# Defaults / Synapse IDs +# ---------------------------------------------------------------------------- +_rna_env = os.environ.get("CNF_RNA_COHORT_SYN_IDS", "") +RNA_TPM_COHORT_SYN_IDS: list[str] = ( + [s.strip() for s in _rna_env.split(",") if s.strip()] + if _rna_env + else ["syn66352931", "syn70765053"] +) + +GLOBAL_PROT_SYN_ID = os.environ.get("CNF_GLOBAL_PROT_SYN_ID", "syn74815895") +PHOSPHO_SYN_ID = os.environ.get("CNF_PHOSPHO_SYN_ID", "syn70078415") + +# Protocol-optimization samples to drop from RNA. +DROP_SAMPLE_SUBSTRINGS: list[str] = [ + "cNF_organoid_DIA_G_02_11Feb25", + "cNF_organoid_DIA_G_05_11Feb25", + "cNF_organoid_DIA_G_06_11Feb25", + "cNF_organoid_DIA_P_02_29Jan25", + "cNF_organoid_DIA_P_05_11Feb25", + "cNF_organoid_DIA_P_06_11Feb25", +] + +STUDY = "cnf" +DEFAULT_OUT_RNA = "/tmp/cnf_transcriptomics.csv" +DEFAULT_OUT_PROT = "/tmp/cnf_proteomics.csv" +DEFAULT_OUT_PHOSPHO = "/tmp/cnf_phosphoproteomics.csv" + + +def configure_logging(): + logging.basicConfig( + format="[%(asctime)s] [%(levelname)s] %(message)s", + datefmt="%Y-%m-%d %H:%M:%S", + level=logging.INFO, + ) + + +# ---------------------------------------------------------------------------- +# Reference data loaders +# ---------------------------------------------------------------------------- +def load_gene_map(genes_path: str) -> pd.DataFrame: + """Load genes.csv. Required columns: gene_symbol, entrez_id.""" + df = pd.read_csv(genes_path) + needed = {"gene_symbol", "entrez_id"} + missing = needed - set(df.columns) + if missing: + raise KeyError(f"genes file missing columns: {missing}") + df = df.dropna(subset=["entrez_id", "gene_symbol"]).copy() + df["entrez_id"] = df["entrez_id"].astype(int) + df = df.drop_duplicates(subset="gene_symbol") + logging.info("Loaded %d gene_symbol → entrez_id mappings", len(df)) + return df[["gene_symbol", "entrez_id"]] + + +def load_sample_map(samples_path: str) -> pd.DataFrame: + """Load cnf_samples.csv; return (other_id, improve_sample_id) frame.""" + df = pd.read_csv(samples_path) + df = df[["other_id", "improve_sample_id"]].drop_duplicates() + df["improve_sample_id"] = df["improve_sample_id"].astype(int) + logging.info("Loaded %d sample mappings", len(df)) + return df + + +def load_phosphosite_map(phosphosites_path: str) -> pd.DataFrame: + """Load phosphosites.csv; return (other_id, phosphosite_id) frame.""" + df = pd.read_csv(phosphosites_path) + needed = {"other_id", "phosphosite_id"} + missing = needed - set(df.columns) + if missing: + raise KeyError(f"phosphosites file missing columns: {missing}") + df = df[["other_id", "phosphosite_id"]].drop_duplicates() + df["phosphosite_id"] = df["phosphosite_id"].astype(int) + logging.info("Loaded %d phosphosite mappings", len(df)) + return df + + +def fetch_synapse_table(syn, syn_id: str) -> pd.DataFrame: + """Download a Synapse file and read it as CSV/TSV.""" + if not syn_id: + raise ValueError("Synapse ID is empty") + logging.info("Fetching %s from Synapse", syn_id) + entity = syn.get(syn_id) + path = entity.path + if path.endswith(".tsv") or path.endswith(".tsv.gz"): + return pd.read_csv(path, sep="\t") + return pd.read_csv(path) + + +# ---------------------------------------------------------------------------- +# Shared helpers +# ---------------------------------------------------------------------------- +def _col_stem(col: str) -> str: + """Return the basename without extension for a path-like column name.""" + base = re.sub(r"^.*[/\\]", "", col) + return re.sub(r"\.[^.]+$", "", base) + + +def _drop_protocol_cols(cols: list[str]) -> list[str]: + """Remove columns whose stem matches any DROP_SAMPLE_SUBSTRINGS entry.""" + kept = [] + for c in cols: + stem = _col_stem(c) + if any(sub in stem for sub in DROP_SAMPLE_SUBSTRINGS): + logging.info(" Dropping protocol optimization sample: %s", stem) + else: + kept.append(c) + return kept + + +def _normalize_col(df: pd.DataFrame, candidates: list[str], target: str) -> pd.DataFrame: + """Rename the first matching candidate column to target.""" + for c in candidates: + if c in df.columns: + return df.rename(columns={c: target}) + raise KeyError( + f"None of {candidates} found in columns {list(df.columns)[:20]}" + ) + + +# ---------------------------------------------------------------------------- +# Transcriptomics +# ---------------------------------------------------------------------------- +def fetch_wide_tpm_cohort(syn, syn_id: str) -> pd.DataFrame: + """Fetch a wide salmon.merged.gene_tpm.tsv and return long-format DataFrame. + + nf-core salmon output: gene_id (Ensembl), gene_name (symbol), then one + column per specimen. Returns (gene_symbol, specimen, transcriptomics, source). + """ + logging.info("Fetching cohort TPM matrix %s", syn_id) + entity = syn.get(syn_id) + path = entity.path + df = pd.read_csv(path, sep="\t") + + gene_col = None + for cname in ["gene_name", "gene_symbol", "Symbol", "GeneSymbol", "gene"]: + if cname in df.columns: + gene_col = cname + break + if gene_col is None: + gene_col = df.columns[0] + logging.warning( + "%s: no gene_name column found; using %s — most rows may fail the gene_map join", + syn_id, gene_col, + ) + + meta_cols = {"gene_id", "gene_name", "gene_symbol", "Symbol", + "GeneSymbol", "gene", "transcript_id"} + sample_cols = [c for c in df.columns if c not in meta_cols] + sample_cols = _drop_protocol_cols(sample_cols) + + if not sample_cols: + raise ValueError( + f"{syn_id}: no specimen columns found after excluding metadata " + f"and protocol optimization samples" + ) + + long = ( + df[[gene_col] + sample_cols] + .rename(columns={gene_col: "gene_symbol"}) + .melt(id_vars="gene_symbol", var_name="specimen", value_name="transcriptomics") + ) + long["source"] = "Synapse" + logging.info( + " %s: %d genes × %d specimens → %d long rows", + syn_id, df[gene_col].nunique(), len(sample_cols), len(long), + ) + return long + + +def build_transcriptomics( + syn, gene_map: pd.DataFrame, sample_map: pd.DataFrame +) -> pd.DataFrame: + """Build the long-format transcriptomics table from per-cohort TPM matrices.""" + if not RNA_TPM_COHORT_SYN_IDS: + logging.error("RNA_TPM_COHORT_SYN_IDS is empty; skipping transcriptomics.") + return pd.DataFrame( + columns=["entrez_id", "improve_sample_id", "source", "study", "transcriptomics"] + ) + + cohort_frames = [] + for syn_id in RNA_TPM_COHORT_SYN_IDS: + try: + cohort_frames.append(fetch_wide_tpm_cohort(syn, syn_id)) + except Exception as exc: + logging.warning("Skipping RNA cohort %s: %s", syn_id, exc) + + if not cohort_frames: + logging.error("All RNA cohort fetches failed; returning empty transcriptomics.") + return pd.DataFrame( + columns=["entrez_id", "improve_sample_id", "source", "study", "transcriptomics"] + ) + + raw = pd.concat(cohort_frames, ignore_index=True) + logging.info("Combined %d RNA cohorts → %d total rows", len(cohort_frames), len(raw)) + + n_before = len(raw) + raw = canonicalize_specimen_column(raw, "specimen") + logging.info( + "Transcriptomics: canonicalized %d → %d rows (%d dropped as unrecognized specimens)", + n_before, len(raw), n_before - len(raw), + ) + + df = ( + raw.merge(gene_map, on="gene_symbol", how="inner") + .merge(sample_map, left_on="specimen", right_on="other_id", how="inner") + ) + df["study"] = STUDY + df["transcriptomics"] = pd.to_numeric(df["transcriptomics"], errors="coerce") + df = df.dropna(subset=["transcriptomics"]) + df = df[["entrez_id", "improve_sample_id", "source", "study", "transcriptomics"]] + + logging.info( + "Built transcriptomics table: %d rows (%d samples × %d genes)", + len(df), df["improve_sample_id"].nunique(), df["entrez_id"].nunique(), + ) + return df + + +# ---------------------------------------------------------------------------- +# Proteomics +# ---------------------------------------------------------------------------- +def build_proteomics( + syn, gene_map: pd.DataFrame, sample_map: pd.DataFrame +) -> pd.DataFrame: + """Build the long-format proteomics table from the batch-corrected Synapse file.""" + raw = fetch_synapse_table(syn, GLOBAL_PROT_SYN_ID) + raw = _normalize_col(raw, + ["Specimen", "specimenID", "specimen_id", "specimen", "specimen_norm", "sample_id"], + "specimen") + raw = _normalize_col(raw, + ["gene_symbol", "Gene", "GeneSymbol", "Symbol", "gene", "feature_id"], + "gene_symbol") + + n_before = len(raw) + raw = canonicalize_specimen_column(raw, "specimen") + logging.info( + "Proteomics: canonicalized %d → %d rows (%d dropped as unrecognized specimens)", + n_before, len(raw), n_before - len(raw), + ) + + raw = _normalize_col(raw, + ["correctedAbundance", "log_ratio", "logRatio", "proteomics", "value"], + "proteomics") + + df = ( + raw.merge(gene_map, on="gene_symbol", how="inner") + .merge(sample_map, left_on="specimen", right_on="other_id", how="inner") + ) + df["source"] = "Synapse" + df["study"] = STUDY + df["proteomics"] = pd.to_numeric(df["proteomics"], errors="coerce") + df = df.dropna(subset=["proteomics"]) + df = df[["entrez_id", "improve_sample_id", "source", "study", "proteomics"]] + + logging.info( + "Built proteomics table: %d rows (%d samples × %d genes)", + len(df), df["improve_sample_id"].nunique(), df["entrez_id"].nunique(), + ) + return df + + +# ---------------------------------------------------------------------------- +# Phosphoproteomics +# ---------------------------------------------------------------------------- +def build_phosphoproteomics( + syn, phosphosite_map: pd.DataFrame, sample_map: pd.DataFrame +) -> pd.DataFrame: + """Build the long-format phosphoproteomics table from the batch-corrected Synapse file.""" + raw = fetch_synapse_table(syn, PHOSPHO_SYN_ID) + raw = _normalize_col(raw, + ["Specimen", "specimenID", "specimen_id", "specimen"], + "specimen") + raw = _normalize_col(raw, + ["site", "Site", "phosphosite", "feature_id"], + "site") + + n_before = len(raw) + raw = canonicalize_specimen_column(raw, "specimen") + logging.info( + "Phosphoproteomics: canonicalized %d → %d rows (%d dropped as unrecognized specimens)", + n_before, len(raw), n_before - len(raw), + ) + + raw = _normalize_col(raw, + ["correctedAbundance", "log_ratio", "logRatio", "phosphoproteomics", "value"], + "phosphoproteomics") + + df = ( + raw.merge(phosphosite_map, left_on="site", right_on="other_id", how="inner") + .merge(sample_map, left_on="specimen", right_on="other_id", how="inner") + ) + df["source"] = "Synapse" + df["study"] = STUDY + df["phosphoproteomics"] = pd.to_numeric(df["phosphoproteomics"], errors="coerce") + df = df.dropna(subset=["phosphoproteomics"]) + df = df[["phosphosite_id", "improve_sample_id", "source", "study", "phosphoproteomics"]] + + logging.info( + "Built phosphoproteomics table: %d rows (%d samples × %d sites)", + len(df), df["improve_sample_id"].nunique(), df["phosphosite_id"].nunique(), + ) + return df + + +# ---------------------------------------------------------------------------- +# Main +# ---------------------------------------------------------------------------- +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("genes", help="Path to genes.csv (gene_symbol → entrez_id)") + parser.add_argument("samples", help="Path to cnf_samples.csv from build_samples") + parser.add_argument("phosphosites", help="Path to phosphosites.csv (other_id → phosphosite_id)") + parser.add_argument("--out_rna", default=DEFAULT_OUT_RNA) + parser.add_argument("--out_prot", default=DEFAULT_OUT_PROT) + parser.add_argument("--out_phospho", default=DEFAULT_OUT_PHOSPHO) + args = parser.parse_args() + + configure_logging() + + gene_map = load_gene_map(args.genes) + sample_map = load_sample_map(args.samples) + phosphosite_map = load_phosphosite_map(args.phosphosites) + + syn = synapseclient.Synapse() + syn.login() + + rna = build_transcriptomics(syn, gene_map, sample_map) + prot = build_proteomics(syn, gene_map, sample_map) + phospho = build_phosphoproteomics(syn, phosphosite_map, sample_map) + + for path in (args.out_rna, args.out_prot, args.out_phospho): + os.makedirs(os.path.dirname(path) or ".", exist_ok=True) + + rna.to_csv(args.out_rna, index=False) + prot.to_csv(args.out_prot, index=False) + phospho.to_csv(args.out_phospho, index=False) + logging.info("Wrote %s (%d rows)", args.out_rna, len(rna)) + logging.info("Wrote %s (%d rows)", args.out_prot, len(prot)) + logging.info("Wrote %s (%d rows)", args.out_phospho, len(phospho)) + + +if __name__ == "__main__": + sys.exit(main() or 0) diff --git a/coderbuild/cnf/README.md b/coderbuild/cnf/README.md new file mode 100644 index 00000000..0755a952 --- /dev/null +++ b/coderbuild/cnf/README.md @@ -0,0 +1,169 @@ +# Cutaneous Neurofibroma (cnf) Organoid Drug Screen + +A coderdata dataset of cutaneous neurofibroma (cNF) patient-derived organoids +with drug-response measurements and matched multi-omic profiling. + +## Overview + +Cutaneous neurofibromas are benign nerve-sheath tumors that develop in patients +with neurofibromatosis type 1 (NF1). This dataset contains drug-screen +viability measurements for 238 small-molecule compounds across 23 cNF tumor +specimens from 10 patients, paired with bulk RNA-seq, global LC-MS/MS +proteomics, and batch-corrected phosphoproteomics for many of the same +specimens. + +| Property | Value | +|---|---| +| Cancer type | Cutaneous Neurofibroma | +| Model type | patient derived organoid | +| Patients | 10 | +| Specimens | 23 (untreated organoids; additional tissue and treated organoid samples also present) | +| Drugs | 238 | +| Modalities included | Transcriptomics (TPM), Proteomics (log ratio), Phosphoproteomics (log ratio) | + +## Multi-tumor patient structure + +Each patient (`NF0017`, `NF0018`, ...) contributed up to three independent +tumor specimens, named `NFxxxx_T1`, `NFxxxx_T2`, `NFxxxx_T3`. This dataset +treats each specimen as an independent sample (its own `improve_sample_id`). +The patient ID is preserved in the `other_names` column so downstream models +can group specimens by patient when needed (e.g. patient-level holdout). + +``` +improve_sample_id other_id common_name other_names other_id_source +2401 NF0019_T1 NF0019_T1 NF0019 NF1_cNF_Project +2402 NF0019_T2 NF0019_T2 NF0019 NF1_cNF_Project +2403 NF0019_T3 NF0019_T3 NF0019 NF1_cNF_Project +``` + +The sample table also includes matched normal skin (`NFxxxx_skin`), primary +tumor tissue (`NFxxxx_Tx_tissue`), and drug-treated organoids +(`NFxxxx_Tx_`). Each gets its own `improve_sample_id` and +`model_type`. Organoids (untreated) receive the lowest IDs, sorted by +specimen ID, to make drug-modelable samples easy to filter. + + +## Drug-response metrics + +The cNF screen mixes two designs: + +- **Multi-dose drugs**: a subset of drugs were screened at multiple + concentrations. These are run through `coderbuild/utils/fit_curve.py` and + contribute the standard schema metrics (`fit_auc`, `fit_ic50`, `fit_einf`, + `fit_hs`, `fit_r2`). + +- **Single-dose drugs**: remaining drugs were screened at a single 1 μM + concentration. These are recorded with `dose_response_metric = uM_viability` + and `dose_response_value = Viability_percentage / 100` (range 0–1, where + 1.0 = full viability and 0.0 = full kill). + +`uM_viability` is a value in the `ResponseMetric` enum. See the Schema +dependency section below. + +## SPECIMEN_DUAL_MAPPINGS edge case + +One viability file covers the treated organoid `NF0021_T1_Onalespid_1uM` but +its drug-response measurements should also be attributed to the untreated +`NF0021_T1`. The `SPECIMEN_DUAL_MAPPINGS` dict in `04-experiments-cnf.py` +handles this by duplicating the rows for every additional specimen listed: + +```python +SPECIMEN_DUAL_MAPPINGS = { + "NF0021_T1_Onalespid_1uM": ["NF0021_T1"], +} +``` + +Add new entries here if additional treated organoids need the same dual +attribution. Both specimens must already exist in `cnf_samples.csv` or the +duplicate rows will be dropped during the sample-map join. + +## Protocol optimization samples + +Six RNA columns in the cohort TPM matrices correspond to protocol optimization +runs rather than real specimens. They are excluded in `02-omics-cnf.py` before +the wide-to-long melt. Column stems matching any of these substrings are +dropped: + +```python +DROP_SAMPLE_SUBSTRINGS = [ + "cNF_organoid_DIA_G_02_11Feb25", + "cNF_organoid_DIA_G_05_11Feb25", + "cNF_organoid_DIA_G_06_11Feb25", + "cNF_organoid_DIA_P_02_29Jan25", + "cNF_organoid_DIA_P_05_11Feb25", + "cNF_organoid_DIA_P_06_11Feb25", +] +``` + +These are matched on the column stem (basename without extension), so both +plain paths and filename columns are handled correctly. If new protocol +optimization runs appear in future cohort data, add their column stems to +this list. + +## Phosphosites dependency and known unmatched sites + +`02-omics-cnf.py` requires `phosphosites.csv` as a positional argument. +The phosphosites reference must be built before the cNF omics step runs. +`build_dataset.py` and `build_all.py` enforce this sequencing. + +The phosphoproteomics output file (`cnf_phosphoproteomics.csv`) is produced +by an inner join between the raw Synapse phospho data and `phosphosites.csv`. +Sites that don't appear in the reference are silently dropped. As of the +current release, **12 sites** in the raw cNF phospho data (syn70078415) are +permanently unmatched because their gene symbols (`BAP18`, `C11orf96`, +`C14orf93`, `C1orf21`, `C2orf49`, etc). The remaining 2,974 of 2,986 unique sites +(99.6%) are present in the output. + +## Data sources + +| Modality | Synapse ID | Notes | +|---|---|---| +| Drug-screen file index | `syn51301431` (table) | `dataType='drug screen'` | +| Drug-screen file parents | `syn51301414`, `syn51301420`, `syn51301426` | One folder per cohort | +| RNA-seq TPM (Cohort 1) | `syn66352931` | `salmon.merged.gene_tpm.tsv` | +| RNA-seq TPM (Cohort 2) | `syn70765053` | `salmon.merged.gene_tpm.tsv` | +| Global proteomics | `syn74815895` | Batch-corrected; `correctedAbundance` column | +| Phospho-proteomics | `syn70078415` | Batch-corrected; `correctedAbundance` column; mapped via `phosphosites.csv` | +| Sample discovery (RNA) | `syn71333780` | All-conditions RNA file for sample enumeration only | +| Normal Skin folder | `syn74284682` | One child per patient for skin sample discovery | + +## Build pipeline + +The build follows coderdata conventions: four shell scripts wrapping Python +implementations, plus shared utilities, packaged in a Docker image. + +``` +coderbuild/cnf/ +├── 01-samples-cnf.py # mint improve_sample_ids, write cnf_samples.csv +├── build_samples.sh +├── 02-omics-cnf.py # write cnf_transcriptomics.csv, cnf_proteomics.csv, +├── build_omics.sh # cnf_phosphoproteomics.csv +├── 03-drugs-cnf.py # call pubchem_retrieval + descriptor utility +├── build_drugs.sh +├── 04-experiments-cnf.py # split single/multi-dose, run fit_curve, write cnf_experiments.tsv +├── build_exp.sh +├── cnf_utils.py # shared specimen canonicalization utilities +├── requirements.txt +└── README.md (this file) + +coderbuild/docker/ +└── Dockerfile.cnf +``` + +### Local build + +```bash +# From repository root +python coderbuild/build_dataset.py --build --validate --dataset cnf +``` + +In the full coderdata build, this dataset is orchestrated by `build_dataset.py` +/ `build_all.py`, which ensures `genes.csv`, `samples.csv`, `drugs.tsv`, and +`phosphosites.csv` are all present before starting the cNF containers. + + +## What's intentionally not included + +- **Mutations** - not yet generated for this dataset. +- **Copy number** - not yet generated for this dataset. + diff --git a/coderbuild/cnf/build_omics.sh b/coderbuild/cnf/build_omics.sh new file mode 100644 index 00000000..7db0d062 --- /dev/null +++ b/coderbuild/cnf/build_omics.sh @@ -0,0 +1,20 @@ +#!/usr/bin/env bash +# build_omics.sh - wraps 02-omics-cnf.py per coderdata convention. +# +# Usage: +# build_omics.sh +# +# - : genes.csv with gene_symbol → entrez_id mappings +# - : cnf_samples.csv produced by build_samples.sh +# - : phosphosites.csv (other_id → phosphosite_id) + +set -euo pipefail + +GENES="${1:?Usage: build_omics.sh }" +SAMPLES="${2:?Usage: build_omics.sh }" +PHOSPHOSITES="${3:?Usage: build_omics.sh }" + +python 02-omics-cnf.py "$GENES" "$SAMPLES" "$PHOSPHOSITES" \ + --out_rna /tmp/cnf_transcriptomics.csv \ + --out_prot /tmp/cnf_proteomics.csv \ + --out_phospho /tmp/cnf_phosphoproteomics.csv \ No newline at end of file diff --git a/coderbuild/cnf/build_samples.sh b/coderbuild/cnf/build_samples.sh new file mode 100644 index 00000000..37af2162 --- /dev/null +++ b/coderbuild/cnf/build_samples.sh @@ -0,0 +1,15 @@ +#!/usr/bin/env bash +# build_samples.sh - wraps 01-samples-cnf.py per coderdata convention. +# +# Usage: +# build_samples.sh +# +# The previous samples file is the latest samples CSV from the coderdata +# build (with the highest improve_sample_id used so far). 01-samples-cnf.py +# reads it, finds the max ID, and increments from there. + +set -euo pipefail + +PREV_SAMPLES="${1:?Usage: build_samples.sh }" + +python 01-samples-cnf.py "$PREV_SAMPLES" --output /tmp/cnf_samples.csv \ No newline at end of file diff --git a/coderbuild/cnf/cnf_utils.py b/coderbuild/cnf/cnf_utils.py new file mode 100644 index 00000000..10ec38d2 --- /dev/null +++ b/coderbuild/cnf/cnf_utils.py @@ -0,0 +1,117 @@ +#!/usr/bin/env python3 +""" +cnf_utils.py: shared specimen classification utilities for the cNF build scripts. + +Used by 01-samples-cnf.py, 02-omics-cnf.py, and 04-experiments-cnf.py to +canonicalize specimen strings so omics data, drug measurements, and the sample +table all use the same improve_sample_id keys. +""" + +import re + + +# ---------------------------------------------------------------------------- +# Sample type → coderdata model_type enum +# ---------------------------------------------------------------------------- +MODEL_TYPE_MAP = { + "organoid": "patient derived organoid", + "treated_organoid": "patient derived organoid", + "tumor_tissue": "tumor", + "normal_skin": "normal_tissue", +} + + + + +# ---------------------------------------------------------------------------- +# Regex constants +# ---------------------------------------------------------------------------- +# Bare tumor organoid: NFxxxx_Tx (4 patient digits, T + tumor digits) +TUMOR_RE = re.compile(r"^NF\d{4}_T\d+$") +# Patient prefix at the start of any specimen string +PATIENT_PREFIX_RE = re.compile(r"^(NF\d{4})") +# Normal skin marker: NFxxxx[_-]skin..., allowing trailing date/replicate junk +SKIN_RE = re.compile(r"^(NF\d{4})[_-]skin", re.IGNORECASE) +# Primary tumor tissue (not organoid culture): NFxxxx_Tx_tissue +TISSUE_RE = re.compile(r"^(NF\d{4}_T\d+)_tissue$", re.IGNORECASE) +# Treated organoid: NFxxxx_Tx_, e.g. NF0021_T1_Onalespid_1uM +TREATED_RE = re.compile(r"^NF\d{4}_T\d+_.+$") + + +# ---------------------------------------------------------------------------- +# Classification +# ---------------------------------------------------------------------------- +def classify_specimen(spec) -> tuple[str, str] | None: + """Classify a raw specimen string into (canonical_id, sample_type) or None. + + sample_type ∈ {"organoid", "treated_organoid", "tumor_tissue", "normal_skin"}. + + Canonical IDs: + organoid → NFxxxx_Tx (suffix stripped) + treated_organoid → NFxxxx_Tx_ (label preserved so + treated samples don't collide + with the untreated organoid) + tumor_tissue → NFxxxx_Tx_tissue (suffix preserved) + normal_skin → NFxxxx_skin (replicates collapsed per patient) + + Unrecognized prefixes and malformed inputs return None. + """ + if spec is None: + return None + s = str(spec).strip() + if not s or s.lower() == "nan": + return None + + # Convert dot-separated form (e.g. RNA file's "NF0017.T1.organoid") to + # underscore-separated. + s_norm = s.replace(".", "_") + + # Skin: any string starting with NFxxxx that contains a "skin" marker + skin_m = SKIN_RE.match(s_norm) + if skin_m: + return (f"{skin_m.group(1)}_skin", "normal_skin") + + # Tumor tissue: NFxxxx_Tx_tissue + tissue_m = TISSUE_RE.match(s_norm) + if tissue_m: + return (f"{tissue_m.group(1)}_tissue", "tumor_tissue") + + # Strip organoid suffix (handles "_organoid" and "_organoids") + s_stripped = re.sub(r"_organoids?$", "", s_norm, flags=re.IGNORECASE) + + # Bare organoid (untreated): NFxxxx_Tx exactly + if TUMOR_RE.match(s_stripped): + return (s_stripped, "organoid") + + # Treated organoid: NFxxxx_Tx_ + if TREATED_RE.match(s_stripped): + return (s_stripped, "treated_organoid") + + return None + + +def patient_from_specimen(specimen: str) -> str: + """Extract NFxxxx patient ID from a canonical specimen ID.""" + m = PATIENT_PREFIX_RE.match(specimen) + return m.group(1) if m else "" + + +def canonicalize_specimen_column(df, specimen_col: str): + """Apply classify_specimen to a dataframe's specimen column in place. + + Returns the same df with: + - rows whose specimen classifies dropped if None + - specimen_col values replaced with their canonical IDs + - a new column 'sample_type' added + + Used by build_omics.py and build_exp.py to map raw Synapse specimen + strings (e.g. "NF0018.T1.organoid", "NF0021.T1.Onalespid.1uM") to + canonical IDs that match the cnf_samples.csv `other_id` column. + """ + classifications = df[specimen_col].apply(classify_specimen) + keep = classifications.notna() + df = df[keep].copy() + classifications = classifications[keep] + df[specimen_col] = [c[0] for c in classifications] + df["sample_type"] = [c[1] for c in classifications] + return df diff --git a/coderbuild/cnf/requirements.txt b/coderbuild/cnf/requirements.txt new file mode 100644 index 00000000..a5677837 --- /dev/null +++ b/coderbuild/cnf/requirements.txt @@ -0,0 +1,9 @@ +pandas>=2.0 +numpy>=1.24,<2.0 +synapseclient>=4.0 +requests>=2.28 +scipy>=1.10 +matplotlib +scikit-learn +scipy +mordred \ No newline at end of file diff --git a/coderbuild/docker/Dockerfile.cnf b/coderbuild/docker/Dockerfile.cnf new file mode 100644 index 00000000..f4a99784 --- /dev/null +++ b/coderbuild/docker/Dockerfile.cnf @@ -0,0 +1,74 @@ +# Dockerfile.cnf - container for building the cNF (cutaneous neurofibroma) +# organoid drug-response dataset for inclusion in coderdata. +# +# Conventions (matching coderbuild/docker/Dockerfile.): +# * Working directory is /coderbuild/cnf +# * Standard utilities are mounted/copied to /coderbuild/utils (fit_curve.py, +# pubchem_retrieval.py, build_drug_desc.py) +# * Synapse auth via SYNAPSE_AUTH_TOKEN env var (passed at `docker run`) +# * Build scripts are invoked from build_dataset.py / build_all.py via the +# standard build_*.sh entry points +# +# Build: +# docker build -f coderbuild/docker/Dockerfile.cnf -t coderdata-cnf . +# +# Run (orchestrated by build_dataset.py / build_all.py): +# docker run --rm \ +# -e SYNAPSE_AUTH_TOKEN=$SYNAPSE_AUTH_TOKEN \ +# -e CNF_RNA_TPM_SYN_ID=$CNF_RNA_TPM_SYN_ID \ +# -v $(pwd)/local_data:/coderbuild/local_data \ +# coderdata-cnf bash build_samples.sh /coderbuild/local_data/prev_samples.csv + +FROM python:3.11-slim + +# Keep scratch files OFF the /tmp bind mount. +# +# build_all.py bind-mounts the host's local/ directory at /tmp in every +# container. Python 3.14 changed the default multiprocessing start method on +# Linux from "fork" to "forkserver", which opens a Unix domain socket under +# tempfile.gettempdir() -- i.e. /tmp, i.e. the bind mount. Docker's macOS file +# sharing does not support chmod() on a socket there, so every Pool worker +# fails with: +# +# OSError: [Errno 22] Invalid argument: '/tmp/pymp-XXXX/sock-YYYY' +# +# This broke build_drug_desc.py and fit_curve.py on 2026-09-03, each at the end +# of a multi-hour step, and it is deterministic so retries do not help. +# +# TMPDIR only affects tempfile.gettempdir() (Python) and tempdir() (R). Explicit +# /tmp/ paths -- which is how every build OUTPUT is written -- still land +# on the bind mount as before. Verified. +RUN mkdir -p /opt/mp_tmp +ENV TMPDIR=/opt/mp_tmp + + +# System deps (rdkit-pypi pulls in some C libs; basic build tools for scipy) +RUN apt-get update && apt-get install -y --no-install-recommends \ + build-essential wget ca-certificates \ + && rm -rf /var/lib/apt/lists/* + +# Set up coderbuild layout +WORKDIR /coderbuild/cnf + +# Install Python deps. rdkit is brought in via the descriptor utility but +# pinned here so the descriptor step is reproducible. +COPY coderbuild/cnf/requirements.txt /coderbuild/cnf/requirements.txt +RUN pip install --no-cache-dir -r requirements.txt rdkit-pypi + +# Copy the cNF build scripts +COPY coderbuild/cnf/01-samples-cnf.py /coderbuild/cnf/ +COPY coderbuild/cnf/02-omics-cnf.py /coderbuild/cnf/ +COPY coderbuild/cnf/03-drugs-cnf.py /coderbuild/cnf/ +COPY coderbuild/cnf/04-experiments-cnf.py /coderbuild/cnf/ +COPY coderbuild/cnf/build_samples.sh /coderbuild/cnf/ +COPY coderbuild/cnf/build_omics.sh /coderbuild/cnf/ +COPY coderbuild/cnf/build_drugs.sh /coderbuild/cnf/ +COPY coderbuild/cnf/build_exp.sh /coderbuild/cnf/ +COPY coderbuild/cnf/cnf_utils.py /coderbuild/cnf/ +COPY coderbuild/utils/build_drug_desc.py /coderbuild/cnf/ + +RUN chmod +x /coderbuild/cnf/*.sh + +# Standard coderdata utilities (fit_curve.py, pubchem_retrieval.py, +COPY coderbuild/utils /coderbuild/utils +