diff --git a/docs/README.md b/docs/README.md index e341b2f5..4be16c0e 100644 --- a/docs/README.md +++ b/docs/README.md @@ -65,9 +65,10 @@ To update pages navigate to `docs/source/`, all .rst and .md files live here. He | Tutorial | docsite_tutorial.ipynb | Deep Learning Tutorial | | Contributing | contribution_guide.rst *includes contribution.md and add_code_guide.md* | Contribution Guide | -**Note:** -- When adding pages, reStructuredText files are the most sphinx friendly and allow for use of helpful directives. However, Markdown files can be used and will render properly when the myst-parser extension is used. -- New pages must be added to the `.. toctree::` in the `index.rst` to be recognized by Sphinx. +**Note:** +- When adding pages, reStructuredText files are the most sphinx friendly and allow for use of helpful directives. However, Markdown files can be used and will render properly when the myst-parser extension is used. +- New pages must be added to the `.. toctree::` in the `index.rst` to be recognized by Sphinx. +- The counts on the Datasets page (`datasets_included.rst`) are generated directly from a local CoderData build using `scripts/gen_dataset_stats.py`, which writes a machine-readable summary to `docs/source/_static/dataset_counts.csv`. To refresh the numbers for a new release, run `python scripts/gen_dataset_stats.py --build-dir --output docs/source/_static/dataset_counts.csv` and update the tables in `datasets_included.rst` accordingly. - Currently the API reference does not include documentation for `plot_2D_respones_metric`, `format`, `split_train_test_validate`, and `split_train_other`. It seems that the most updated `coderdata` directory currently does not include docstrings for these functions. ### Tutorial diff --git a/docs/source/_static/dataset_counts.csv b/docs/source/_static/dataset_counts.csv new file mode 100644 index 00000000..0905b013 --- /dev/null +++ b/docs/source/_static/dataset_counts.csv @@ -0,0 +1,19 @@ +dataset,model_type,samples,drugs,genes,experiments,transcriptomics,proteomics,phosphoproteomics,mutations,copy_number,drug_descriptors,combinations,dose_response_metrics +beataml,ex vivo,1022,164,19221,324980,13573491,1437831,0,7233,0,264696,0,aac;auc;dss;fit_auc;fit_ec50;fit_ec50se;fit_einf;fit_hs;fit_ic50;fit_r2 +bladder,patient derived organoid;tumor;patient derived xenograft;xenograft derived organoid,134,50,19403,33000,815975,0,0,3568,1135661,80700,0,aac;auc;dss;fit_auc;fit_ec50;fit_ec50se;fit_einf;fit_hs;fit_ic50;fit_r2 +ccle,cell line,503,24,19380,115430,9114448,2180178,0,224974,6788934,38736,0,aac;auc;dss;fit_auc;fit_ec50;fit_ec50se;fit_einf;fit_hs;fit_ic50;fit_r2 +colorectal,patient derived organoid;tumor,61,10,20839,1400,156993,0,0,44634,689155,16140,0,aac;auc;dss;fit_auc;fit_ec50;fit_ec50se;fit_einf;fit_hs;fit_ic50;fit_r2 +cptac,tumor,1139,0,19278,0,21170373,9486498,0,252398,19261183,0,0, +ctrpv2,cell line,848,460,19381,3101330,15739656,3011948,0,389773,10906812,742440,0,aac;auc;dss;fit_auc;fit_ec50;fit_ec50se;fit_einf;fit_hs;fit_ic50;fit_r2 +fimm,cell line,53,52,19353,26630,919104,250970,0,12743,760509,83928,0,aac;auc;dss;fit_auc;fit_ec50;fit_ec50se;fit_einf;fit_hs;fit_ic50;fit_r2 +gcsi,cell line,571,43,19380,132290,10148440,2247800,0,276904,7197012,69402,0,aac;auc;dss;fit_auc;fit_ec50;fit_ec50se;fit_einf;fit_hs;fit_ic50;fit_r2 +gdscv1,cell line,985,291,19158,2454890,18501510,4694495,0,3385316,37266972,469674,0,aac;auc;dss;fit_auc;fit_ec50;fit_ec50se;fit_einf;fit_hs;fit_ic50;fit_r2 +gdscv2,cell line,807,170,19158,1146910,15192913,3879361,0,2852337,30512750,274380,0,aac;auc;dss;fit_auc;fit_ec50;fit_ec50se;fit_einf;fit_hs;fit_ic50;fit_r2 +hcmi,patient derived organoid,1087,0,19218,0,12540435,0,0,100602,6044025,0,0, +liver,patient derived organoid,62,76,27163,44530,1063454,452662,0,12810,1686901,122664,0,aac;auc;dss;fit_auc;fit_ec50;fit_ec50se;fit_einf;fit_hs;fit_ic50;fit_r2 +mpnst,3D-MEDS;patient derived xenograft;tumor,47,39,19731,27985,1612748,52638,0,84720,541907,62946,75,aac;abc;auc;delta_auc;dss;fit_auc;fit_ec50;fit_ec50se;fit_einf;fit_hs;fit_ic50;fit_r2;lmm;mrecist;tgi +nci60,cell line,84,23267,19368,12364040,1340360,387298,0,54400,1094391,37551324,0,aac;auc;dss;fit_auc;fit_ec50;fit_ec50se;fit_einf;fit_hs;fit_ic50;fit_r2 +novartis,patient derived xenograft,348,25,19042,6950,6215451,0,0,106692,6219933,40350,0,abc;lmm;mrecist;tgi +pancreatic,patient derived organoid,49,25,19070,1900,1031014,0,0,78589,955688,40350,0,aac;auc;dss;fit_auc;fit_ec50;fit_ec50se;fit_einf;fit_hs;fit_ic50;fit_r2 +prism,cell line,479,1418,19381,6385340,9076152,2144440,0,249890,6992973,2288652,0,aac;auc;dss;fit_auc;fit_ec50;fit_ec50se;fit_einf;fit_hs;fit_ic50;fit_r2 +sarcoma,tumor;patient derived organoid,36,34,17895,275,573198,0,0,92,0,54876,0,published_auc diff --git a/docs/source/_static/dataset_summary_statistics.csv b/docs/source/_static/dataset_summary_statistics.csv index 7dab0f42..867b7f37 100644 --- a/docs/source/_static/dataset_summary_statistics.csv +++ b/docs/source/_static/dataset_summary_statistics.csv @@ -1,19 +1,19 @@ -dataset,sample_drug_pairs,sample_drug_transcript_pairs,sample_drug_transcriptomics_mutation_pairs,sample_drug_transcriptomics_copynumber_pairs,sample_drug_mutation_copynumber_pairs -beataml,31926.0,4137.0,3958.0,, -bladder,3300.0,840.0,640.0,640.0,3100.0 -ccle,11543.0,10887.0,10792.0,10887.0,11118.0 -colorectal,140.0,60.0,60.0,60.0,140.0 -cptac,,,,, -ctrpv2,309401.0,300507.0,295742.0,299698.0,300616.0 -fimm,2663.0,2457.0,2457.0,2457.0,2611.0 -gcsi,13398.0,12506.0,12338.0,12506.0,13112.0 -gdscv1,247753.0,245220.0,241999.0,241240.0,242570.0 -gdscv2,115440.0,114373.0,112829.0,112523.0,113133.0 -hcmi,,,,, -liver,4453.0,4453.0,4453.0,4453.0,4453.0 -mpnst,272.0,193.0,184.0,191.0,184.0 -nci60,2960756.0,2329149.0,2329132.0,2329149.0,2784474.0 -novartis,1766.0,1734.0,1734.0,1723.0,1723.0 -pancreatic,190.0,190.0,185.0,185.0,185.0 -prism,638983.0,632078.0,630672.0,632078.0,636226.0 -sarcoma,275.0,234.0,187.0,, +dataset,sample_drug_pairs,sample_drug_transcript_pairs,sample_drug_transcriptomics_mutation_pairs,sample_drug_transcriptomics_copynumber_pairs,sample_drug_mutation_copynumber_pairs +beataml,31926,4137,3958,, +bladder,3300,840,640,640,3100 +ccle,11543,10887,10792,10887,11118 +colorectal,140,60,60,60,140 +cptac,,,,, +ctrpv2,309401,300507,295742,299698,300616 +fimm,2663,2457,2457,2457,2611 +gcsi,13398,12506,12338,12506,13112 +gdscv1,247753,245220,241999,241240,242570 +gdscv2,115440,114373,112829,112523,113133 +hcmi,,,,, +liver,4453,4453,4453,4453,4453 +mpnst,272,193,184,191,184 +nci60,2960756,2329149,2329132,2329149,2784474 +novartis,1766,1734,1734,1723,1723 +pancreatic,190,190,185,185,185 +prism,638983,632078,630672,632078,636226 +sarcoma,275,234,187,, diff --git a/docs/source/datasets_included.rst b/docs/source/datasets_included.rst index e505edec..dc416d8e 100644 --- a/docs/source/datasets_included.rst +++ b/docs/source/datasets_included.rst @@ -1,45 +1,46 @@ Datasets Included ================= -This page provides an overview of the datasets included in CoderData version 2.2.0. This package collects 18 diverse sets of paired molecular datasets with corresponding drug sensitivity data. All data here is reprocessed and standardized so it can be easily used as a benchmark dataset for drug response prediction machine learning models. +This page provides an overview of the datasets included in CoderData version 2.4. This package collects 18 diverse sets of paired molecular datasets with corresponding drug sensitivity data. All data here is reprocessed and standardized so it can be easily used as a benchmark dataset for drug response prediction machine learning models. The dataset files are in csv format and are available at the link below: -Figshare record: https://api.figshare.com/v2/articles/28823159 +Figshare record: https://api.figshare.com/v2/articles/33990850 -Version: 2.2.0 +Version: 2.4 --------------------------- Dataset Overview --------------------------- .. csv-table:: Datasets and Modalities - :header: "Dataset", "References", "Sample", "Drug", "Drug Descriptor", "Experiments", "Transcriptomics", "Proteomics", "Mutations", "Copy Number" - :widths: 14, 12, 6, 8, 15, 12, 12, 12, 12, 12 - - "BeatAML", "[1]_, [2]_", "1022", "164", "X", "X", "X", "X", "X", "" - "Bladder", "[3]_", "134", "50", "X", "X", "X", "", "X", "X" - "CCLE", "[4]_", "502", "24", "X", "X", "X", "X", "X", "X" - "Colorectal ", "[18]_", "61", "10", "X", "", "X", "", "X", "X" - "CPTAC", "[5]_", "1139", "", "", "", "X", "X", "X", "X" - "CTRPv2", "[6]_, [7]_, [8]_", "846", "459", "X", "X", "X", "", "X", "X" - "FIMM", "[9]_, [10]_", "52", "52", "X", "X", "X", "", "", "" - "GDSC v1", "[23]_, [24]_, [25]_", "984", "294", "X", "", "X", "X", "X", "X" - "GDSC v2", "[23]_, [24]_, [25]_", "806", "171", "X", "", "X", "X", "X", "X" - "gCSI", "[21]_, [22]_", "569", "44", "X", "", "X", "X", "X", "X" - "HCMI", "[11]_", "886", "", "", "", "X", "", "X", "X" - "Liver", "[19]_", "62", "76", "X", "", "X", "", "X", "X" - "MPNST", "[12]_", "50", "30", "X", "X", "X", "X", "X", "X" - "NCI60", "[13]_", "83", "55157", "X", "X", "X", "X", "X", "" - "Novartis", "[20]_", "386", "25", "X", "", "X", "", "X", "X" - "Pancreatic", "[14]_", "70", "25", "X", "X", "X", "", "X", "X" - "PRISM", "[15]_, [16]_", "478", "1419", "X", "X", "X", "", "", "" - "Sarcoma", "[17]_", "36", "34", "X", "X", "X", "", "X", "" - - -The table above lists the datasets included in CoderData version 2.2.0, along with references to their original publications, counts of samples and drugs, and the types of data available for each dataset. + :header: "Dataset", "References", "Model Type", "Sample", "Drug", "Drug Descriptor", "Experiments", "Transcriptomics", "Proteomics", "Mutations", "Copy Number" + :widths: 12, 10, 16, 6, 7, 12, 10, 10, 10, 10, 10 + + "BeatAML", "[1]_, [2]_", "ex vivo", "1022", "164", "X", "X", "X", "X", "X", "" + "Bladder", "[3]_", "patient derived organoid; patient derived xenograft; tumor; xenograft derived organoid", "134", "50", "X", "X", "X", "", "X", "X" + "CCLE", "[4]_", "cell line", "503", "24", "X", "X", "X", "X", "X", "X" + "Colorectal", "[18]_", "patient derived organoid; tumor", "61", "10", "X", "X", "X", "", "X", "X" + "CPTAC", "[5]_", "tumor", "1139", "", "", "", "X", "X", "X", "X" + "CTRPv2", "[6]_, [7]_, [8]_", "cell line", "848", "460", "X", "X", "X", "X", "X", "X" + "FIMM", "[9]_, [10]_", "cell line", "53", "52", "X", "X", "X", "X", "X", "X" + "GDSC v1", "[23]_, [24]_, [25]_", "cell line", "985", "291", "X", "X", "X", "X", "X", "X" + "GDSC v2", "[23]_, [24]_, [25]_", "cell line", "807", "170", "X", "X", "X", "X", "X", "X" + "gCSI", "[21]_, [22]_", "cell line", "571", "43", "X", "X", "X", "X", "X", "X" + "HCMI", "[11]_", "patient derived organoid", "1087", "", "", "", "X", "", "X", "X" + "Liver", "[19]_", "patient derived organoid", "62", "76", "X", "X", "X", "X", "X", "X" + "MPNST", "[12]_", "3D-MEDS; patient derived xenograft; tumor", "47", "39", "X", "X", "X", "X", "X", "X" + "NCI60", "[13]_", "cell line", "84", "23267", "X", "X", "X", "X", "X", "X" + "Novartis", "[20]_", "patient derived xenograft", "348", "25", "X", "X", "X", "", "X", "X" + "Pancreatic", "[14]_", "patient derived organoid", "49", "25", "X", "X", "X", "", "X", "X" + "PRISM", "[15]_, [16]_", "cell line", "479", "1418", "X", "X", "X", "X", "X", "X" + "Sarcoma", "[17]_", "tumor; patient derived organoid", "36", "34", "X", "X", "X", "", "X", "" + + +The table above lists the datasets included in CoderData version 2.4, along with references to their original publications, counts of samples and drugs, and the types of data available for each dataset. CoderData includes the following data: +- Model Type - the biological model the samples derive from (cell line, patient derived organoid, patient derived xenograft, tumor, ex vivo, or 3D-MEDS microtissue) - Sample - cell lines, patient-derived samples, or patient-derived organoids - Drug - compounds tested for sensitivity - Drug Descriptor - molecular descriptors for each drug (computed using RDKit) diff --git a/scripts/gen_dataset_stats.py b/scripts/gen_dataset_stats.py new file mode 100644 index 00000000..346ea843 --- /dev/null +++ b/scripts/gen_dataset_stats.py @@ -0,0 +1,208 @@ +#!/usr/bin/env python3 +"""Generate a comprehensive per-dataset counts table from a local CoderData build. + +Reads a build output directory (the ``all_files_dir`` produced by +``coderbuild/build_all.py``) and writes a single wide CSV summarizing every +dataset: unique samples, unique drugs, unique genes, per-modality row counts, +and the set of dose-response metrics present. The output is written to the +docs static directory so the documentation site can render current numbers. + +This script reads the build files directly (no network, no installed +``coderdata`` package required), so it can be run against any local build. + +Usage: + python scripts/gen_dataset_stats.py \ + --build-dir local/all_files_dir \ + --output docs/source/_static/dataset_counts.csv +""" + +import argparse +import csv +import gzip +import io +import os +from collections import OrderedDict + + +# Modalities whose presence/row-count we report, in display order. +OMICS = ["transcriptomics", "proteomics", "phosphoproteomics", + "mutations", "copy_number"] +DRUGISH = ["drugs", "drug_descriptors", "experiments", "combinations"] + +# Column that identifies the unique key for count-distinct on each file type. +UNIQUE_KEY = { + "samples": "improve_sample_id", + "drugs": "improve_drug_id", +} +# Files that use TSV rather than CSV. +TSV_TYPES = {"drugs", "drug_descriptors", "experiments", "combinations"} +# Gene-bearing omics used for the unique-gene tally. +GENE_TYPES = ["transcriptomics", "proteomics", "mutations", "copy_number"] + + +def _open_any(path): + """Open a plain or .gz text file for reading.""" + if path.endswith(".gz"): + return io.TextIOWrapper(gzip.open(path, "rb"), encoding="utf-8") + return open(path, "r", encoding="utf-8") + + +def _find_file(build_dir, dataset, modality): + """Return the path to _.(csv|tsv)[.gz] if present.""" + for ext in (".csv", ".tsv", ".csv.gz", ".tsv.gz"): + p = os.path.join(build_dir, f"{dataset}_{modality}{ext}") + if os.path.exists(p): + return p + return None + + +def _delimiter(path): + return "\t" if (".tsv" in path) else "," + + +def _header_index(path, column): + """Return (delimiter, column_index) for a file, or (delim, None) if absent. + + Reads only the first line. Uses fast raw splitting rather than csv module. + """ + delim = _delimiter(path) + with _open_any(path) as f: + header = f.readline().rstrip("\n").rstrip("\r") + cols = header.split(delim) + try: + return delim, cols.index(column) + except ValueError: + return delim, None + + +def _count_rows(path): + """Number of data rows (excludes header). Fast raw line count.""" + with _open_any(path) as f: + n = sum(1 for _ in f) + return max(n - 1, 0) + + +def _iter_column(path, column): + """Yield the value of one column per data row using fast line splitting. + + Assumes the target column contains no embedded delimiter (true for the + id/metric columns we tally here). Skips the header. + """ + delim, idx = _header_index(path, column) + if idx is None: + return + with _open_any(path) as f: + f.readline() # skip header + for line in f: + parts = line.rstrip("\n").rstrip("\r").split(delim) + if idx < len(parts): + v = parts[idx] + if v and v != "NA": + yield v + + +def _count_unique(path, column): + """Number of unique non-empty values in a column (fast path).""" + return len(set(_iter_column(path, column))) + + +def _unique_genes(build_dir, dataset): + """Unique entrez_id across all gene-bearing omics for a dataset.""" + genes = set() + for modality in GENE_TYPES: + p = _find_file(build_dir, dataset, modality) + if not p: + continue + genes.update(_iter_column(p, "entrez_id")) + return len(genes) + + +def _dose_response_metrics(build_dir, dataset): + """Sorted set of dose_response_metric values in the experiments file.""" + p = _find_file(build_dir, dataset, "experiments") + if not p: + return [] + return sorted(set(_iter_column(p, "dose_response_metric"))) + + +def _model_types(build_dir, dataset): + """Distinct model_type values in the samples file, ordered by frequency.""" + p = _find_file(build_dir, dataset, "samples") + if not p: + return [] + counts = {} + for v in _iter_column(p, "model_type"): + counts[v] = counts.get(v, 0) + 1 + return [k for k, _ in sorted(counts.items(), key=lambda kv: (-kv[1], kv[0]))] + + +def discover_datasets(build_dir): + """Dataset names inferred from _samples.csv files.""" + names = set() + for fn in os.listdir(build_dir): + if fn.endswith("_samples.csv"): + names.add(fn[: -len("_samples.csv")]) + return sorted(names) + + +def build_rows(build_dir, datasets): + rows = [] + for d in datasets: + rec = OrderedDict() + rec["dataset"] = d + rec["model_type"] = ";".join(_model_types(build_dir, d)) + + samples_path = _find_file(build_dir, d, "samples") + rec["samples"] = _count_unique(samples_path, "improve_sample_id") if samples_path else 0 + + drugs_path = _find_file(build_dir, d, "drugs") + rec["drugs"] = _count_unique(drugs_path, "improve_drug_id") if drugs_path else 0 + + rec["genes"] = _unique_genes(build_dir, d) + + exp_path = _find_file(build_dir, d, "experiments") + rec["experiments"] = _count_rows(exp_path) if exp_path else 0 + + # per-modality presence (X / blank) and row counts + for modality in OMICS + ["drug_descriptors", "combinations"]: + p = _find_file(build_dir, d, modality) + rec[modality] = _count_rows(p) if p else 0 + + rec["dose_response_metrics"] = ";".join(_dose_response_metrics(build_dir, d)) + rows.append(rec) + return rows + + +def main(): + ap = argparse.ArgumentParser(description=__doc__, + formatter_class=argparse.RawDescriptionHelpFormatter) + ap.add_argument("--build-dir", default="local/all_files_dir", + help="build output directory (default: local/all_files_dir)") + ap.add_argument("--output", default="docs/source/_static/dataset_counts.csv", + help="output CSV path (default: docs/source/_static/dataset_counts.csv)") + ap.add_argument("--datasets", nargs="*", default=None, + help="restrict to these dataset names (default: all discovered)") + args = ap.parse_args() + + if not os.path.isdir(args.build_dir): + raise SystemExit(f"build dir not found: {args.build_dir}") + + datasets = args.datasets or discover_datasets(args.build_dir) + if not datasets: + raise SystemExit(f"no *_samples.csv found in {args.build_dir}") + + rows = build_rows(args.build_dir, datasets) + + os.makedirs(os.path.dirname(args.output) or ".", exist_ok=True) + fieldnames = list(rows[0].keys()) + with open(args.output, "w", newline="", encoding="utf-8") as f: + writer = csv.DictWriter(f, fieldnames=fieldnames) + writer.writeheader() + writer.writerows(rows) + + print(f"Wrote {args.output} with {len(rows)} datasets and {len(fieldnames)} columns.") + print("Datasets:", ", ".join(datasets)) + + +if __name__ == "__main__": + main()