Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
7 changes: 4 additions & 3 deletions docs/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -65,9 +65,10 @@ To update pages navigate to `docs/source/`, all .rst and .md files live here. He
| Tutorial | docsite_tutorial.ipynb | Deep Learning Tutorial |
| Contributing | contribution_guide.rst *includes contribution.md and add_code_guide.md* | Contribution Guide |

**Note:**
- When adding pages, reStructuredText files are the most sphinx friendly and allow for use of helpful directives. However, Markdown files can be used and will render properly when the myst-parser extension is used.
- New pages must be added to the `.. toctree::` in the `index.rst` to be recognized by Sphinx.
**Note:**
- When adding pages, reStructuredText files are the most sphinx friendly and allow for use of helpful directives. However, Markdown files can be used and will render properly when the myst-parser extension is used.
- New pages must be added to the `.. toctree::` in the `index.rst` to be recognized by Sphinx.
- The counts on the Datasets page (`datasets_included.rst`) are generated directly from a local CoderData build using `scripts/gen_dataset_stats.py`, which writes a machine-readable summary to `docs/source/_static/dataset_counts.csv`. To refresh the numbers for a new release, run `python scripts/gen_dataset_stats.py --build-dir <build_output_dir> --output docs/source/_static/dataset_counts.csv` and update the tables in `datasets_included.rst` accordingly.
- Currently the API reference does not include documentation for `plot_2D_respones_metric`, `format`, `split_train_test_validate`, and `split_train_other`. It seems that the most updated `coderdata` directory currently does not include docstrings for these functions.

### Tutorial
Expand Down
19 changes: 19 additions & 0 deletions docs/source/_static/dataset_counts.csv
Original file line number Diff line number Diff line change
@@ -0,0 +1,19 @@
dataset,model_type,samples,drugs,genes,experiments,transcriptomics,proteomics,phosphoproteomics,mutations,copy_number,drug_descriptors,combinations,dose_response_metrics
beataml,ex vivo,1022,164,19221,324980,13573491,1437831,0,7233,0,264696,0,aac;auc;dss;fit_auc;fit_ec50;fit_ec50se;fit_einf;fit_hs;fit_ic50;fit_r2
bladder,patient derived organoid;tumor;patient derived xenograft;xenograft derived organoid,134,50,19403,33000,815975,0,0,3568,1135661,80700,0,aac;auc;dss;fit_auc;fit_ec50;fit_ec50se;fit_einf;fit_hs;fit_ic50;fit_r2
ccle,cell line,503,24,19380,115430,9114448,2180178,0,224974,6788934,38736,0,aac;auc;dss;fit_auc;fit_ec50;fit_ec50se;fit_einf;fit_hs;fit_ic50;fit_r2
colorectal,patient derived organoid;tumor,61,10,20839,1400,156993,0,0,44634,689155,16140,0,aac;auc;dss;fit_auc;fit_ec50;fit_ec50se;fit_einf;fit_hs;fit_ic50;fit_r2
cptac,tumor,1139,0,19278,0,21170373,9486498,0,252398,19261183,0,0,
ctrpv2,cell line,848,460,19381,3101330,15739656,3011948,0,389773,10906812,742440,0,aac;auc;dss;fit_auc;fit_ec50;fit_ec50se;fit_einf;fit_hs;fit_ic50;fit_r2
fimm,cell line,53,52,19353,26630,919104,250970,0,12743,760509,83928,0,aac;auc;dss;fit_auc;fit_ec50;fit_ec50se;fit_einf;fit_hs;fit_ic50;fit_r2
gcsi,cell line,571,43,19380,132290,10148440,2247800,0,276904,7197012,69402,0,aac;auc;dss;fit_auc;fit_ec50;fit_ec50se;fit_einf;fit_hs;fit_ic50;fit_r2
gdscv1,cell line,985,291,19158,2454890,18501510,4694495,0,3385316,37266972,469674,0,aac;auc;dss;fit_auc;fit_ec50;fit_ec50se;fit_einf;fit_hs;fit_ic50;fit_r2
gdscv2,cell line,807,170,19158,1146910,15192913,3879361,0,2852337,30512750,274380,0,aac;auc;dss;fit_auc;fit_ec50;fit_ec50se;fit_einf;fit_hs;fit_ic50;fit_r2
hcmi,patient derived organoid,1087,0,19218,0,12540435,0,0,100602,6044025,0,0,
liver,patient derived organoid,62,76,27163,44530,1063454,452662,0,12810,1686901,122664,0,aac;auc;dss;fit_auc;fit_ec50;fit_ec50se;fit_einf;fit_hs;fit_ic50;fit_r2
mpnst,3D-MEDS;patient derived xenograft;tumor,47,39,19731,27985,1612748,52638,0,84720,541907,62946,75,aac;abc;auc;delta_auc;dss;fit_auc;fit_ec50;fit_ec50se;fit_einf;fit_hs;fit_ic50;fit_r2;lmm;mrecist;tgi
nci60,cell line,84,23267,19368,12364040,1340360,387298,0,54400,1094391,37551324,0,aac;auc;dss;fit_auc;fit_ec50;fit_ec50se;fit_einf;fit_hs;fit_ic50;fit_r2
novartis,patient derived xenograft,348,25,19042,6950,6215451,0,0,106692,6219933,40350,0,abc;lmm;mrecist;tgi
pancreatic,patient derived organoid,49,25,19070,1900,1031014,0,0,78589,955688,40350,0,aac;auc;dss;fit_auc;fit_ec50;fit_ec50se;fit_einf;fit_hs;fit_ic50;fit_r2
prism,cell line,479,1418,19381,6385340,9076152,2144440,0,249890,6992973,2288652,0,aac;auc;dss;fit_auc;fit_ec50;fit_ec50se;fit_einf;fit_hs;fit_ic50;fit_r2
sarcoma,tumor;patient derived organoid,36,34,17895,275,573198,0,0,92,0,54876,0,published_auc
38 changes: 19 additions & 19 deletions docs/source/_static/dataset_summary_statistics.csv
Original file line number Diff line number Diff line change
@@ -1,19 +1,19 @@
dataset,sample_drug_pairs,sample_drug_transcript_pairs,sample_drug_transcriptomics_mutation_pairs,sample_drug_transcriptomics_copynumber_pairs,sample_drug_mutation_copynumber_pairs
beataml,31926.0,4137.0,3958.0,,
bladder,3300.0,840.0,640.0,640.0,3100.0
ccle,11543.0,10887.0,10792.0,10887.0,11118.0
colorectal,140.0,60.0,60.0,60.0,140.0
cptac,,,,,
ctrpv2,309401.0,300507.0,295742.0,299698.0,300616.0
fimm,2663.0,2457.0,2457.0,2457.0,2611.0
gcsi,13398.0,12506.0,12338.0,12506.0,13112.0
gdscv1,247753.0,245220.0,241999.0,241240.0,242570.0
gdscv2,115440.0,114373.0,112829.0,112523.0,113133.0
hcmi,,,,,
liver,4453.0,4453.0,4453.0,4453.0,4453.0
mpnst,272.0,193.0,184.0,191.0,184.0
nci60,2960756.0,2329149.0,2329132.0,2329149.0,2784474.0
novartis,1766.0,1734.0,1734.0,1723.0,1723.0
pancreatic,190.0,190.0,185.0,185.0,185.0
prism,638983.0,632078.0,630672.0,632078.0,636226.0
sarcoma,275.0,234.0,187.0,,
dataset,sample_drug_pairs,sample_drug_transcript_pairs,sample_drug_transcriptomics_mutation_pairs,sample_drug_transcriptomics_copynumber_pairs,sample_drug_mutation_copynumber_pairs
beataml,31926,4137,3958,,
bladder,3300,840,640,640,3100
ccle,11543,10887,10792,10887,11118
colorectal,140,60,60,60,140
cptac,,,,,
ctrpv2,309401,300507,295742,299698,300616
fimm,2663,2457,2457,2457,2611
gcsi,13398,12506,12338,12506,13112
gdscv1,247753,245220,241999,241240,242570
gdscv2,115440,114373,112829,112523,113133
hcmi,,,,,
liver,4453,4453,4453,4453,4453
mpnst,272,193,184,191,184
nci60,2960756,2329149,2329132,2329149,2784474
novartis,1766,1734,1734,1723,1723
pancreatic,190,190,185,185,185
prism,638983,632078,630672,632078,636226
sarcoma,275,234,187,,
55 changes: 28 additions & 27 deletions docs/source/datasets_included.rst
Original file line number Diff line number Diff line change
@@ -1,45 +1,46 @@
Datasets Included
=================

This page provides an overview of the datasets included in CoderData version 2.2.0. This package collects 18 diverse sets of paired molecular datasets with corresponding drug sensitivity data. All data here is reprocessed and standardized so it can be easily used as a benchmark dataset for drug response prediction machine learning models.
This page provides an overview of the datasets included in CoderData version 2.4. This package collects 18 diverse sets of paired molecular datasets with corresponding drug sensitivity data. All data here is reprocessed and standardized so it can be easily used as a benchmark dataset for drug response prediction machine learning models.

The dataset files are in csv format and are available at the link below:

Figshare record: https://api.figshare.com/v2/articles/28823159
Figshare record: https://api.figshare.com/v2/articles/33990850

Version: 2.2.0
Version: 2.4

---------------------------
Dataset Overview
---------------------------
.. csv-table:: Datasets and Modalities
:header: "Dataset", "References", "Sample", "Drug", "Drug Descriptor", "Experiments", "Transcriptomics", "Proteomics", "Mutations", "Copy Number"
:widths: 14, 12, 6, 8, 15, 12, 12, 12, 12, 12

"BeatAML", "[1]_, [2]_", "1022", "164", "X", "X", "X", "X", "X", ""
"Bladder", "[3]_", "134", "50", "X", "X", "X", "", "X", "X"
"CCLE", "[4]_", "502", "24", "X", "X", "X", "X", "X", "X"
"Colorectal ", "[18]_", "61", "10", "X", "", "X", "", "X", "X"
"CPTAC", "[5]_", "1139", "", "", "", "X", "X", "X", "X"
"CTRPv2", "[6]_, [7]_, [8]_", "846", "459", "X", "X", "X", "", "X", "X"
"FIMM", "[9]_, [10]_", "52", "52", "X", "X", "X", "", "", ""
"GDSC v1", "[23]_, [24]_, [25]_", "984", "294", "X", "", "X", "X", "X", "X"
"GDSC v2", "[23]_, [24]_, [25]_", "806", "171", "X", "", "X", "X", "X", "X"
"gCSI", "[21]_, [22]_", "569", "44", "X", "", "X", "X", "X", "X"
"HCMI", "[11]_", "886", "", "", "", "X", "", "X", "X"
"Liver", "[19]_", "62", "76", "X", "", "X", "", "X", "X"
"MPNST", "[12]_", "50", "30", "X", "X", "X", "X", "X", "X"
"NCI60", "[13]_", "83", "55157", "X", "X", "X", "X", "X", ""
"Novartis", "[20]_", "386", "25", "X", "", "X", "", "X", "X"
"Pancreatic", "[14]_", "70", "25", "X", "X", "X", "", "X", "X"
"PRISM", "[15]_, [16]_", "478", "1419", "X", "X", "X", "", "", ""
"Sarcoma", "[17]_", "36", "34", "X", "X", "X", "", "X", ""


The table above lists the datasets included in CoderData version 2.2.0, along with references to their original publications, counts of samples and drugs, and the types of data available for each dataset.
:header: "Dataset", "References", "Model Type", "Sample", "Drug", "Drug Descriptor", "Experiments", "Transcriptomics", "Proteomics", "Mutations", "Copy Number"
:widths: 12, 10, 16, 6, 7, 12, 10, 10, 10, 10, 10

"BeatAML", "[1]_, [2]_", "ex vivo", "1022", "164", "X", "X", "X", "X", "X", ""
"Bladder", "[3]_", "patient derived organoid; patient derived xenograft; tumor; xenograft derived organoid", "134", "50", "X", "X", "X", "", "X", "X"
"CCLE", "[4]_", "cell line", "503", "24", "X", "X", "X", "X", "X", "X"
"Colorectal", "[18]_", "patient derived organoid; tumor", "61", "10", "X", "X", "X", "", "X", "X"
"CPTAC", "[5]_", "tumor", "1139", "", "", "", "X", "X", "X", "X"
"CTRPv2", "[6]_, [7]_, [8]_", "cell line", "848", "460", "X", "X", "X", "X", "X", "X"
"FIMM", "[9]_, [10]_", "cell line", "53", "52", "X", "X", "X", "X", "X", "X"
"GDSC v1", "[23]_, [24]_, [25]_", "cell line", "985", "291", "X", "X", "X", "X", "X", "X"
"GDSC v2", "[23]_, [24]_, [25]_", "cell line", "807", "170", "X", "X", "X", "X", "X", "X"
"gCSI", "[21]_, [22]_", "cell line", "571", "43", "X", "X", "X", "X", "X", "X"
"HCMI", "[11]_", "patient derived organoid", "1087", "", "", "", "X", "", "X", "X"
"Liver", "[19]_", "patient derived organoid", "62", "76", "X", "X", "X", "X", "X", "X"
"MPNST", "[12]_", "3D-MEDS; patient derived xenograft; tumor", "47", "39", "X", "X", "X", "X", "X", "X"
"NCI60", "[13]_", "cell line", "84", "23267", "X", "X", "X", "X", "X", "X"
"Novartis", "[20]_", "patient derived xenograft", "348", "25", "X", "X", "X", "", "X", "X"
"Pancreatic", "[14]_", "patient derived organoid", "49", "25", "X", "X", "X", "", "X", "X"
"PRISM", "[15]_, [16]_", "cell line", "479", "1418", "X", "X", "X", "X", "X", "X"
"Sarcoma", "[17]_", "tumor; patient derived organoid", "36", "34", "X", "X", "X", "", "X", ""


The table above lists the datasets included in CoderData version 2.4, along with references to their original publications, counts of samples and drugs, and the types of data available for each dataset.

CoderData includes the following data:

- Model Type - the biological model the samples derive from (cell line, patient derived organoid, patient derived xenograft, tumor, ex vivo, or 3D-MEDS microtissue)
- Sample - cell lines, patient-derived samples, or patient-derived organoids
- Drug - compounds tested for sensitivity
- Drug Descriptor - molecular descriptors for each drug (computed using RDKit)
Expand Down
208 changes: 208 additions & 0 deletions scripts/gen_dataset_stats.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,208 @@
#!/usr/bin/env python3
"""Generate a comprehensive per-dataset counts table from a local CoderData build.

Reads a build output directory (the ``all_files_dir`` produced by
``coderbuild/build_all.py``) and writes a single wide CSV summarizing every
dataset: unique samples, unique drugs, unique genes, per-modality row counts,
and the set of dose-response metrics present. The output is written to the
docs static directory so the documentation site can render current numbers.

This script reads the build files directly (no network, no installed
``coderdata`` package required), so it can be run against any local build.

Usage:
python scripts/gen_dataset_stats.py \
--build-dir local/all_files_dir \
--output docs/source/_static/dataset_counts.csv
"""

import argparse
import csv
import gzip
import io
import os
from collections import OrderedDict


# Modalities whose presence/row-count we report, in display order.
OMICS = ["transcriptomics", "proteomics", "phosphoproteomics",
"mutations", "copy_number"]
DRUGISH = ["drugs", "drug_descriptors", "experiments", "combinations"]

# Column that identifies the unique key for count-distinct on each file type.
UNIQUE_KEY = {
"samples": "improve_sample_id",
"drugs": "improve_drug_id",
}
# Files that use TSV rather than CSV.
TSV_TYPES = {"drugs", "drug_descriptors", "experiments", "combinations"}
# Gene-bearing omics used for the unique-gene tally.
GENE_TYPES = ["transcriptomics", "proteomics", "mutations", "copy_number"]


def _open_any(path):
"""Open a plain or .gz text file for reading."""
if path.endswith(".gz"):
return io.TextIOWrapper(gzip.open(path, "rb"), encoding="utf-8")
return open(path, "r", encoding="utf-8")


def _find_file(build_dir, dataset, modality):
"""Return the path to <dataset>_<modality>.(csv|tsv)[.gz] if present."""
for ext in (".csv", ".tsv", ".csv.gz", ".tsv.gz"):
p = os.path.join(build_dir, f"{dataset}_{modality}{ext}")
if os.path.exists(p):
return p
return None


def _delimiter(path):
return "\t" if (".tsv" in path) else ","


def _header_index(path, column):
"""Return (delimiter, column_index) for a file, or (delim, None) if absent.

Reads only the first line. Uses fast raw splitting rather than csv module.
"""
delim = _delimiter(path)
with _open_any(path) as f:
header = f.readline().rstrip("\n").rstrip("\r")
cols = header.split(delim)
try:
return delim, cols.index(column)
except ValueError:
return delim, None


def _count_rows(path):
"""Number of data rows (excludes header). Fast raw line count."""
with _open_any(path) as f:
n = sum(1 for _ in f)
return max(n - 1, 0)


def _iter_column(path, column):
"""Yield the value of one column per data row using fast line splitting.

Assumes the target column contains no embedded delimiter (true for the
id/metric columns we tally here). Skips the header.
"""
delim, idx = _header_index(path, column)
if idx is None:
return
with _open_any(path) as f:
f.readline() # skip header
for line in f:
parts = line.rstrip("\n").rstrip("\r").split(delim)
if idx < len(parts):
v = parts[idx]
if v and v != "NA":
yield v


def _count_unique(path, column):
"""Number of unique non-empty values in a column (fast path)."""
return len(set(_iter_column(path, column)))


def _unique_genes(build_dir, dataset):
"""Unique entrez_id across all gene-bearing omics for a dataset."""
genes = set()
for modality in GENE_TYPES:
p = _find_file(build_dir, dataset, modality)
if not p:
continue
genes.update(_iter_column(p, "entrez_id"))
return len(genes)


def _dose_response_metrics(build_dir, dataset):
"""Sorted set of dose_response_metric values in the experiments file."""
p = _find_file(build_dir, dataset, "experiments")
if not p:
return []
return sorted(set(_iter_column(p, "dose_response_metric")))


def _model_types(build_dir, dataset):
"""Distinct model_type values in the samples file, ordered by frequency."""
p = _find_file(build_dir, dataset, "samples")
if not p:
return []
counts = {}
for v in _iter_column(p, "model_type"):
counts[v] = counts.get(v, 0) + 1
return [k for k, _ in sorted(counts.items(), key=lambda kv: (-kv[1], kv[0]))]


def discover_datasets(build_dir):
"""Dataset names inferred from <dataset>_samples.csv files."""
names = set()
for fn in os.listdir(build_dir):
if fn.endswith("_samples.csv"):
names.add(fn[: -len("_samples.csv")])
return sorted(names)


def build_rows(build_dir, datasets):
rows = []
for d in datasets:
rec = OrderedDict()
rec["dataset"] = d
rec["model_type"] = ";".join(_model_types(build_dir, d))

samples_path = _find_file(build_dir, d, "samples")
rec["samples"] = _count_unique(samples_path, "improve_sample_id") if samples_path else 0

drugs_path = _find_file(build_dir, d, "drugs")
rec["drugs"] = _count_unique(drugs_path, "improve_drug_id") if drugs_path else 0

rec["genes"] = _unique_genes(build_dir, d)

exp_path = _find_file(build_dir, d, "experiments")
rec["experiments"] = _count_rows(exp_path) if exp_path else 0

# per-modality presence (X / blank) and row counts
for modality in OMICS + ["drug_descriptors", "combinations"]:
p = _find_file(build_dir, d, modality)
rec[modality] = _count_rows(p) if p else 0

rec["dose_response_metrics"] = ";".join(_dose_response_metrics(build_dir, d))
rows.append(rec)
return rows


def main():
ap = argparse.ArgumentParser(description=__doc__,
formatter_class=argparse.RawDescriptionHelpFormatter)
ap.add_argument("--build-dir", default="local/all_files_dir",
help="build output directory (default: local/all_files_dir)")
ap.add_argument("--output", default="docs/source/_static/dataset_counts.csv",
help="output CSV path (default: docs/source/_static/dataset_counts.csv)")
ap.add_argument("--datasets", nargs="*", default=None,
help="restrict to these dataset names (default: all discovered)")
args = ap.parse_args()

if not os.path.isdir(args.build_dir):
raise SystemExit(f"build dir not found: {args.build_dir}")

datasets = args.datasets or discover_datasets(args.build_dir)
if not datasets:
raise SystemExit(f"no *_samples.csv found in {args.build_dir}")

rows = build_rows(args.build_dir, datasets)

os.makedirs(os.path.dirname(args.output) or ".", exist_ok=True)
fieldnames = list(rows[0].keys())
with open(args.output, "w", newline="", encoding="utf-8") as f:
writer = csv.DictWriter(f, fieldnames=fieldnames)
writer.writeheader()
writer.writerows(rows)

print(f"Wrote {args.output} with {len(rows)} datasets and {len(fieldnames)} columns.")
print("Datasets:", ", ".join(datasets))


if __name__ == "__main__":
main()