{"count":4,"items":[{"access":"registration","article":null,"cancer_slugs":["pan-cancer"],"doi":null,"formats":["SVS","MAF","VCF","BAM","TSV","JSON"],"hf":null,"huggingface":null,"kind":"registry","license":"NIH GDS Policy \u2014 open tier; controlled tier via dbGaP","modalities":["genomics","transcriptomics","histopathology","radiology-ct","radiology-mri"],"name":"TCGA \u2014 The Cancer Genome Atlas (via NCI Genomic Data Commons)","page":"/ai-oncology/datasets/tcga","provider":"NCI / NHGRI; hosted by the Genomic Data Commons","size":{"items":30000,"notes_en":"33 cancer types; ~11,000 patients with molecular data; diagnostic slides for most cases; matched clinical follow-up","notes_pl":"33 typy nowotwor\u00f3w; ok. 11 000 pacjent\u00f3w z danymi molekularnymi; preparaty diagnostyczne dla wi\u0119kszo\u015bci przypadk\u00f3w; dopasowana obserwacja kliniczna","patients":11000,"unit":"diagnostic + tissue WSIs"},"slug":"tcga","summary":"The reference multi-omics cancer cohort: molecular profiles, clinical outcomes and whole-slide images for 33 cancer types, downloadable through the GDC portal and API.","tasks":["classification","prognosis","survival-analysis","variant-effect","gene-expression"],"url":"https://portal.gdc.cancer.gov/","verified_at":"2026-09-05T22:25:58.814252"},{"access":"registration","article":null,"cancer_slugs":["pan-cancer"],"doi":null,"formats":["TSV","SVS","DICOM","MAF"],"hf":null,"huggingface":null,"kind":"registry","license":"open (proteomics via PDC) / controlled (raw sequencing via dbGaP)","modalities":["proteomics","genomics","transcriptomics","histopathology","radiology-ct"],"name":"CPTAC \u2014 Clinical Proteomic Tumor Analysis Consortium","page":"/ai-oncology/datasets/cptac","provider":"NCI Office of Cancer Clinical Proteomics Research","size":{"notes_en":"10+ tumour types with proteogenomic profiling (proteome, phosphoproteome) on genomically characterised cases; images in TCIA","notes_pl":"10+ typ\u00f3w guz\u00f3w z profilowaniem proteogenomicznym (proteom, fosfoproteom) na przypadkach scharakteryzowanych genomowo; obrazy w TCIA","patients":1000},"slug":"cptac","summary":"Proteogenomic companion to TCGA: the largest public set where protein-level measurements, genomics and images exist for the same tumours.","tasks":["classification","prognosis","gene-expression"],"url":"https://proteomics.cancer.gov/programs/cptac","verified_at":"2026-09-05T22:25:58.828625"},{"access":"open","article":null,"cancer_slugs":["pan-cancer"],"doi":null,"formats":["CSV","Parquet"],"hf":null,"huggingface":null,"kind":"dataset","license":"CC BY 4.0","modalities":["genomics","transcriptomics","molecules"],"name":"DepMap \u2014 Cancer Dependency Map","page":"/ai-oncology/datasets/depmap","provider":"Broad Institute","size":{"items":1900,"notes_en":"genome-wide CRISPR knockout screens, drug sensitivity (PRISM) and multi-omics per line; quarterly releases","notes_pl":"genomowe screeny CRISPR, wra\u017cliwo\u015b\u0107 na leki (PRISM) i multi-omika per linia; wydania kwartalne","unit":"cancer cell lines"},"slug":"depmap","summary":"Public map of cancer vulnerabilities in cell lines \u2014 the training and validation ground for target-discovery and drug-response models.","tasks":["drug-discovery","gene-expression"],"url":"https://depmap.org/portal/","verified_at":"2026-09-05T22:25:59.011882"},{"access":"open","article":null,"cancer_slugs":[],"doi":null,"formats":["other"],"hf":null,"huggingface":null,"kind":"dataset","license":"Apache-2.0","modalities":["transcriptomics","single-cell"],"name":"Genecorpus-30M","page":"/ai-oncology/datasets/genecorpus-30m","provider":"Theodoris/Ellinor lab (Broad Institute, Massachusetts General Hospital)","size":{"cells":"27 406 217 komorek po filtrach jakosci (z 29 900 531 zebranych)","headline":"~30 mln ludzkich transkryptomow pojedynczych komorek","source_datasets":"561 publicznie dostepnych zbiorow danych"},"slug":"genecorpus-30m","summary":"Genecorpus-30M is a pretraining corpus of about 30 million human single-cell transcriptomes assembled from 561 publicly available datasets, of which 27,406,217 cells passed quality filters. It was built to pretrain Geneformer (Theodoris et al., Nature 2023) and is released under Apache-2.0. Crucially for oncology use: cells with high mutational burden \u2014 malignant cells and immortalized cell lines \u2014 were deliberately excluded from the corpus.","tasks":["gene-expression","single-cell"],"url":"https://huggingface.co/datasets/ctheodoris/Genecorpus-30M","verified_at":"2026-09-08T04:55:51.018757"}]}
