{"count":6,"items":[{"access":"registration","article":null,"cancer_slugs":["pan-cancer"],"doi":null,"formats":["SVS","MAF","VCF","BAM","TSV","JSON"],"hf":null,"huggingface":null,"kind":"registry","license":"NIH GDS Policy \u2014 open tier; controlled tier via dbGaP","modalities":["genomics","transcriptomics","histopathology","radiology-ct","radiology-mri"],"name":"TCGA \u2014 The Cancer Genome Atlas (via NCI Genomic Data Commons)","page":"/ai-oncology/datasets/tcga","provider":"NCI / NHGRI; hosted by the Genomic Data Commons","size":{"items":30000,"notes_en":"33 cancer types; ~11,000 patients with molecular data; diagnostic slides for most cases; matched clinical follow-up","notes_pl":"33 typy nowotwor\u00f3w; ok. 11 000 pacjent\u00f3w z danymi molekularnymi; preparaty diagnostyczne dla wi\u0119kszo\u015bci przypadk\u00f3w; dopasowana obserwacja kliniczna","patients":11000,"unit":"diagnostic + tissue WSIs"},"slug":"tcga","summary":"The reference multi-omics cancer cohort: molecular profiles, clinical outcomes and whole-slide images for 33 cancer types, downloadable through the GDC portal and API.","tasks":["classification","prognosis","survival-analysis","variant-effect","gene-expression"],"url":"https://portal.gdc.cancer.gov/","verified_at":"2026-09-05T22:25:58.814252"},{"access":"registration","article":null,"cancer_slugs":["pan-cancer"],"doi":null,"formats":["TSV","SVS","DICOM","MAF"],"hf":null,"huggingface":null,"kind":"registry","license":"open (proteomics via PDC) / controlled (raw sequencing via dbGaP)","modalities":["proteomics","genomics","transcriptomics","histopathology","radiology-ct"],"name":"CPTAC \u2014 Clinical Proteomic Tumor Analysis Consortium","page":"/ai-oncology/datasets/cptac","provider":"NCI Office of Cancer Clinical Proteomics Research","size":{"notes_en":"10+ tumour types with proteogenomic profiling (proteome, phosphoproteome) on genomically characterised cases; images in TCIA","notes_pl":"10+ typ\u00f3w guz\u00f3w z profilowaniem proteogenomicznym (proteom, fosfoproteom) na przypadkach scharakteryzowanych genomowo; obrazy w TCIA","patients":1000},"slug":"cptac","summary":"Proteogenomic companion to TCGA: the largest public set where protein-level measurements, genomics and images exist for the same tumours.","tasks":["classification","prognosis","gene-expression"],"url":"https://proteomics.cancer.gov/programs/cptac","verified_at":"2026-09-05T22:25:58.828625"},{"access":"registration","article":null,"cancer_slugs":["pan-cancer"],"doi":null,"formats":["MAF","TSV"],"hf":null,"huggingface":null,"kind":"registry","license":"AACR GENIE data-use terms (Synapse registration)","modalities":["genomics","ehr"],"name":"AACR Project GENIE","page":"/ai-oncology/datasets/aacr-genie","provider":"American Association for Cancer Research; hosted on Synapse / cBioPortal","size":{"notes_en":"clinical-grade tumour sequencing from 19+ institutions with limited clinical data; releases twice a year","notes_pl":"sekwencjonowanie guz\u00f3w klasy klinicznej z 19+ instytucji z ograniczonymi danymi klinicznymi; wydania dwa razy w roku","patients":200000},"slug":"aacr-genie","summary":"The largest real-world clinical sequencing registry in oncology \u2014 the place to test whether a variant-effect or biomarker model generalises beyond TCGA.","tasks":["variant-effect","prognosis","survival-analysis"],"url":"https://www.aacr.org/professionals/research/aacr-project-genie/","verified_at":"2026-09-05T22:25:58.995187"},{"access":"open","article":null,"cancer_slugs":["pan-cancer"],"doi":null,"formats":["CSV","Parquet"],"hf":null,"huggingface":null,"kind":"dataset","license":"CC BY 4.0","modalities":["genomics","transcriptomics","molecules"],"name":"DepMap \u2014 Cancer Dependency Map","page":"/ai-oncology/datasets/depmap","provider":"Broad Institute","size":{"items":1900,"notes_en":"genome-wide CRISPR knockout screens, drug sensitivity (PRISM) and multi-omics per line; quarterly releases","notes_pl":"genomowe screeny CRISPR, wra\u017cliwo\u015b\u0107 na leki (PRISM) i multi-omika per linia; wydania kwartalne","unit":"cancer cell lines"},"slug":"depmap","summary":"Public map of cancer vulnerabilities in cell lines \u2014 the training and validation ground for target-discovery and drug-response models.","tasks":["drug-discovery","gene-expression"],"url":"https://depmap.org/portal/","verified_at":"2026-09-05T22:25:59.011882"},{"access":"open","article":"/blog/modele-prezentacji-antygenu","cancer_slugs":["pan-cancer"],"doi":null,"formats":["text","CSV"],"hf":null,"huggingface":null,"kind":"registry","license":"CC BY-ND 4.0 (see site terms)","modalities":["genomics","protein-sequence"],"name":"IPD-IMGT/HLA Database","page":"/ai-oncology/datasets/ipd-imgt-hla","provider":"EMBL-EBI / Anthony Nolan Research Institute","size":{"items":30894,"notes_en":"as of June 2026: 9,279 HLA-A, 11,258 HLA-B, 9,416 HLA-C \u2014 while one patient carries at most six","notes_pl":"stan na czerwiec 2026: 9279 HLA-A, 11 258 HLA-B, 9416 HLA-C \u2014 a jeden pacjent ma najwy\u017cej sze\u015b\u0107","unit":"named class I alleles"},"slug":"ipd-imgt-hla","summary":"The naming authority for HLA alleles: the catalogue whose size \u2014 thirty thousand names against roughly a hundred well-measured alleles \u2014 defines the central problem of this field.","tasks":["antigen-presentation"],"url":"https://hla.alleles.org/","verified_at":"2026-09-05T22:25:59.303091"},{"access":"open","article":null,"cancer_slugs":[],"doi":null,"formats":["FASTQ","BAM","VCF"],"hf":null,"huggingface":null,"kind":null,"license":"Materia\u0142 referencyjny NIST; zgoda dawc\u00f3w z PGP na redystrybucj\u0119 komercyjn\u0105","modalities":["genomics"],"name":"Genome in a Bottle (GIAB)","page":"/ai-oncology/datasets/giab-genome-in-a-bottle","provider":"National Institute of Standards and Technology (NIST), USA","size":{"items":7,"notes_en":"Seven characterised genomes: the pilot genome NA12878/HG001 from the HapMap collection and two Personal Genome Project trios \u2014 Ashkenazi Jewish (HG002, HG003, HG004) and Han Chinese (HG005, HG006, HG007).","notes_pl":"Siedem scharakteryzowanych genom\u00f3w: genom pilota\u017cowy NA12878/HG001 z kolekcji HapMap oraz dwa trio z Personal Genome Project \u2014 aszkenazyjskie (HG002, HG003, HG004) i chi\u0144skie Han (HG005, HG006, HG007).","patients":7,"unit":"genomy referencyjne"},"slug":"giab-genome-in-a-bottle","summary":"A public benchmark resource run by NIST within a consortium of public bodies, companies and academic centres. It contains seven thoroughly characterised human genomes with benchmark variant sets and high-confidence regions, and is used to measure how accurately a bioinformatics pipeline calls variants. It is not an oncology dataset and not an imaging dataset \u2014 it is a reference point for genome readout itself. The consortium works with the GA4GH Benchmarking Team on comparison standards. Distribution: the NIST reference material shop, the Coriell Institute and the Personal Genome Project.","tasks":[],"url":"https://www.nist.gov/programs-projects/genome-bottle","verified_at":null}]}
