{"count":16,"items":[{"access":"open","article":null,"cancer_slugs":["pan-cancer"],"doi":null,"formats":["text","JSON","PNG"],"hf":null,"huggingface":null,"kind":"corpus","license":"PubMed metadata public domain; PMC OA articles under CC licences per article","modalities":["literature"],"name":"PubMed / PMC Open Access Subset","page":"/ai-oncology/datasets/pubmed-pmc-oa","provider":"US National Library of Medicine","size":{"items":36000000,"notes_en":"abstracts for all of PubMed; full text and figures for the open-access subset","notes_pl":"abstrakty ca\u0142ego PubMed; pe\u0142ne teksty i ryciny dla podzbioru open-access","unit":"citations (PubMed); millions of full texts in PMC OA"},"slug":"pubmed-pmc-oa","summary":"The literature corpus behind BiomedBERT, BiomedCLIP and CONCH's caption data; the entry point for any oncology NLP or vision-language pretraining.","tasks":["information-extraction","question-answering","image-text-retrieval"],"url":"https://pmc.ncbi.nlm.nih.gov/tools/openftlist/","verified_at":"2026-09-05T22:25:59.060602"},{"access":"open","article":null,"cancer_slugs":["pan-cancer"],"doi":null,"formats":["JSON","CSV"],"hf":null,"huggingface":null,"kind":"registry","license":"public domain","modalities":["clinical-text"],"name":"ClinicalTrials.gov","page":"/ai-oncology/datasets/clinicaltrials-gov","provider":"US National Library of Medicine","size":{"items":500000,"notes_en":"structured eligibility criteria, arms, outcomes and results; full API","notes_pl":"ustrukturyzowane kryteria kwalifikacji, ramiona, punkty ko\u0144cowe i wyniki; pe\u0142ne API","unit":"registered studies"},"slug":"clinicaltrials-gov","summary":"The registry that trial-matching systems such as TrialGPT retrieve from; oncology is its largest therapeutic area.","tasks":["clinical-trial-matching","information-extraction"],"url":"https://clinicaltrials.gov/","verified_at":"2026-09-05T22:25:59.082279"},{"access":"open","article":null,"cancer_slugs":["pan-cancer"],"doi":null,"formats":["JSON"],"hf":null,"huggingface":null,"kind":"benchmark","license":"MIT","modalities":["clinical-text"],"name":"MedQA (USMLE)","page":"/ai-oncology/datasets/medqa","provider":"Jin et al. (Columbia University)","size":{"items":12723,"notes_en":"multiple-choice medical licensing exam questions; also Mandarin and Traditional Chinese subsets","notes_pl":"pytania wielokrotnego wyboru z egzamin\u00f3w lekarskich; tak\u017ce podzbiory w mandary\u0144skim i tradycyjnym chi\u0144skim","unit":"questions (English)"},"slug":"medqa","summary":"The exam-style benchmark on which Med-PaLM 2, GPT-4 and MedGemma report headline accuracy; useful for comparing models, not for judging clinical safety.","tasks":["question-answering"],"url":"https://github.com/jind11/MedQA","verified_at":"2026-09-05T22:25:59.099060"},{"access":"registration","article":null,"cancer_slugs":["pan-cancer"],"doi":null,"formats":["SVS","MAF","VCF","BAM","TSV","JSON"],"hf":null,"huggingface":null,"kind":"registry","license":"NIH GDS Policy \u2014 open tier; controlled tier via dbGaP","modalities":["genomics","transcriptomics","histopathology","radiology-ct","radiology-mri"],"name":"TCGA \u2014 The Cancer Genome Atlas (via NCI Genomic Data Commons)","page":"/ai-oncology/datasets/tcga","provider":"NCI / NHGRI; hosted by the Genomic Data Commons","size":{"items":30000,"notes_en":"33 cancer types; ~11,000 patients with molecular data; diagnostic slides for most cases; matched clinical follow-up","notes_pl":"33 typy nowotwor\u00f3w; ok. 11 000 pacjent\u00f3w z danymi molekularnymi; preparaty diagnostyczne dla wi\u0119kszo\u015bci przypadk\u00f3w; dopasowana obserwacja kliniczna","patients":11000,"unit":"diagnostic + tissue WSIs"},"slug":"tcga","summary":"The reference multi-omics cancer cohort: molecular profiles, clinical outcomes and whole-slide images for 33 cancer types, downloadable through the GDC portal and API.","tasks":["classification","prognosis","survival-analysis","variant-effect","gene-expression"],"url":"https://portal.gdc.cancer.gov/","verified_at":"2026-09-05T22:25:58.814252"},{"access":"registration","article":null,"cancer_slugs":["pan-cancer"],"doi":null,"formats":["TSV","SVS","DICOM","MAF"],"hf":null,"huggingface":null,"kind":"registry","license":"open (proteomics via PDC) / controlled (raw sequencing via dbGaP)","modalities":["proteomics","genomics","transcriptomics","histopathology","radiology-ct"],"name":"CPTAC \u2014 Clinical Proteomic Tumor Analysis Consortium","page":"/ai-oncology/datasets/cptac","provider":"NCI Office of Cancer Clinical Proteomics Research","size":{"notes_en":"10+ tumour types with proteogenomic profiling (proteome, phosphoproteome) on genomically characterised cases; images in TCIA","notes_pl":"10+ typ\u00f3w guz\u00f3w z profilowaniem proteogenomicznym (proteom, fosfoproteom) na przypadkach scharakteryzowanych genomowo; obrazy w TCIA","patients":1000},"slug":"cptac","summary":"Proteogenomic companion to TCGA: the largest public set where protein-level measurements, genomics and images exist for the same tumours.","tasks":["classification","prognosis","gene-expression"],"url":"https://proteomics.cancer.gov/programs/cptac","verified_at":"2026-09-05T22:25:58.828625"},{"access":"open","article":null,"cancer_slugs":["pan-cancer"],"doi":null,"formats":["DICOM","NIfTI","CSV"],"hf":null,"huggingface":null,"kind":"registry","license":"mostly CC BY 3.0/4.0 per collection; some restricted collections","modalities":["radiology-ct","radiology-mri","pet","mammography","histopathology","radiology-xray"],"name":"TCIA \u2014 The Cancer Imaging Archive","page":"/ai-oncology/datasets/tcia","provider":"NCI Cancer Imaging Program; hosted by the University of Arkansas for Medical Sciences","size":{"items":200,"notes_en":"hundreds of collections, tens of thousands of patients; DICOM with linked clinical and sometimes genomic data","notes_pl":"setki kolekcji, dziesi\u0105tki tysi\u0119cy pacjent\u00f3w; DICOM z powi\u0105zanymi danymi klinicznymi, czasem genomowymi","unit":"collections"},"slug":"tcia","summary":"The main public archive of de-identified cancer imaging, organised into collections by disease and modality; the source of most public radiology training data in oncology.","tasks":["segmentation","detection","classification","prognosis"],"url":"https://www.cancerimagingarchive.net/","verified_at":"2026-09-05T22:25:58.836069"},{"access":"registration","article":null,"cancer_slugs":["pan-cancer"],"doi":null,"formats":["MAF","TSV"],"hf":null,"huggingface":null,"kind":"registry","license":"AACR GENIE data-use terms (Synapse registration)","modalities":["genomics","ehr"],"name":"AACR Project GENIE","page":"/ai-oncology/datasets/aacr-genie","provider":"American Association for Cancer Research; hosted on Synapse / cBioPortal","size":{"notes_en":"clinical-grade tumour sequencing from 19+ institutions with limited clinical data; releases twice a year","notes_pl":"sekwencjonowanie guz\u00f3w klasy klinicznej z 19+ instytucji z ograniczonymi danymi klinicznymi; wydania dwa razy w roku","patients":200000},"slug":"aacr-genie","summary":"The largest real-world clinical sequencing registry in oncology \u2014 the place to test whether a variant-effect or biomarker model generalises beyond TCGA.","tasks":["variant-effect","prognosis","survival-analysis"],"url":"https://www.aacr.org/professionals/research/aacr-project-genie/","verified_at":"2026-09-05T22:25:58.995187"},{"access":"open","article":null,"cancer_slugs":["pan-cancer"],"doi":null,"formats":["CSV","Parquet"],"hf":null,"huggingface":null,"kind":"dataset","license":"CC BY 4.0","modalities":["genomics","transcriptomics","molecules"],"name":"DepMap \u2014 Cancer Dependency Map","page":"/ai-oncology/datasets/depmap","provider":"Broad Institute","size":{"items":1900,"notes_en":"genome-wide CRISPR knockout screens, drug sensitivity (PRISM) and multi-omics per line; quarterly releases","notes_pl":"genomowe screeny CRISPR, wra\u017cliwo\u015b\u0107 na leki (PRISM) i multi-omika per linia; wydania kwartalne","unit":"cancer cell lines"},"slug":"depmap","summary":"Public map of cancer vulnerabilities in cell lines \u2014 the training and validation ground for target-discovery and drug-response models.","tasks":["drug-discovery","gene-expression"],"url":"https://depmap.org/portal/","verified_at":"2026-09-05T22:25:59.011882"},{"access":"open","article":null,"cancer_slugs":["pan-cancer"],"doi":null,"formats":["PDB","mmCIF"],"hf":null,"huggingface":null,"kind":"registry","license":"CC0","modalities":["protein-sequence","molecules"],"name":"Protein Data Bank (wwPDB / RCSB)","page":"/ai-oncology/datasets/pdb","provider":"Worldwide Protein Data Bank","size":{"items":220000,"notes_en":"X-ray, cryo-EM and NMR structures of proteins, nucleic acids and complexes; the training ground of AlphaFold","notes_pl":"struktury rentgenowskie, krio-EM i NMR bia\u0142ek, kwas\u00f3w nukleinowych i kompleks\u00f3w; grunt treningowy AlphaFold","unit":"experimental structures"},"slug":"pdb","summary":"The open archive of experimentally solved macromolecular structures, including oncology targets (kinases, KRAS, p53) and their drug complexes.","tasks":["protein-structure","drug-discovery"],"url":"https://www.rcsb.org/","verified_at":"2026-09-05T22:25:59.035915"},{"access":"open","article":"/blog/modele-prezentacji-antygenu","cancer_slugs":["pan-cancer"],"doi":null,"formats":["CSV","JSON"],"hf":null,"huggingface":null,"kind":"registry","license":"free to use; NIAID-funded public resource","modalities":["protein-sequence","immunopeptidomics"],"name":"IEDB \u2014 Immune Epitope Database","page":"/ai-oncology/datasets/iedb","provider":"La Jolla Institute for Immunology, funded by NIAID","size":{"items":1600000,"notes_en":"curated from published literature and direct submissions; includes MHC binding assays, MS-eluted ligands and T-cell assays","notes_pl":"kuratorowane z literatury i zg\u0142osze\u0144 bezpo\u015brednich; zawiera testy wi\u0105zania MHC, ligandy ze spektrometru i testy limfocyt\u00f3w T","unit":"epitope-related records"},"slug":"iedb","summary":"The field's central repository of epitope data and the source of almost every training set for peptide\u2013MHC models \u2014 and of their allele skew.","tasks":["peptide-mhc-binding","antigen-presentation"],"url":"https://www.iedb.org/","verified_at":"2026-09-05T22:25:59.114121"},{"access":"open","article":"/blog/modele-prezentacji-antygenu","cancer_slugs":["pan-cancer"],"doi":null,"formats":["CSV"],"hf":null,"huggingface":null,"kind":"benchmark","license":"public","modalities":["protein-sequence"],"name":"IEDB automated benchmark (MHC class I)","page":"/ai-oncology/datasets/iedb-benchmark","provider":"La Jolla Institute for Immunology","size":{"notes_en":"runs continuously on newly deposited data, before it can leak into anyone's training set","notes_pl":"dzia\u0142a na bie\u017c\u0105co na \u015bwie\u017co deponowanych danych, zanim mog\u0105 trafi\u0107 do czyjegokolwiek zbioru treningowego"},"slug":"iedb-benchmark","summary":"The only prospective, third-party benchmark in the field. Its eight-year summary is sobering: leading methods are statistically indistinguishable, and a new method needs about four years before enough data accumulate to judge it.","tasks":["peptide-mhc-binding","antigen-presentation"],"url":"http://tools.iedb.org/auto_bench/mhci/weekly/","verified_at":"2026-09-05T22:25:59.140589"},{"access":"open","article":"/blog/modele-prezentacji-antygenu","cancer_slugs":["pan-cancer"],"doi":"10.1038/s41587-019-0322-9","formats":["CSV","text"],"hf":null,"huggingface":null,"kind":"dataset","license":"published supplementary data (see papers)","modalities":["immunopeptidomics","protein-sequence"],"name":"Mono-allelic HLA class I peptidome (Sarkizova / Abelin)","page":"/ai-oncology/datasets/monoallelic-peptidome","provider":"Broad Institute / Dana-Farber Cancer Institute","size":{"items":186464,"notes_en":"95 mono-allelic cell lines (31 HLA-A, 40 HLA-B, 21 HLA-C, 3 HLA-G), median 1,860 peptides per allele; the earlier Abelin 2017 set covered 16 alleles and >24,000 peptides","notes_pl":"95 linii monoallelicznych (31 HLA-A, 40 HLA-B, 21 HLA-C, 3 HLA-G), mediana 1860 peptyd\u00f3w na allel; wcze\u015bniejszy zbi\u00f3r Abelin 2017 obj\u0105\u0142 16 alleli i ponad 24 000 peptyd\u00f3w","unit":"peptides"},"slug":"monoallelic-peptidome","summary":"The engineered-cell peptidome that gave the field clean allele labels; fifteen of its alleles had no described motif before, and the panel covers at least one allele in 95% of people worldwide.","tasks":["antigen-presentation","peptide-mhc-binding"],"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC7008090/","verified_at":"2026-09-05T22:25:59.152441"},{"access":"open","article":"/blog/modele-prezentacji-antygenu","cancer_slugs":["pan-cancer"],"doi":null,"formats":["CSV"],"hf":null,"huggingface":null,"kind":"dataset","license":"CC BY 4.0","modalities":["immunopeptidomics"],"name":"HLA Ligand Atlas","page":"/ai-oncology/datasets/hla-ligand-atlas","provider":"University of T\u00fcbingen (Rammensee / Walz groups)","size":{"items":90428,"notes_en":"29 tissue types from 21 donors \u2014 benign tissue, which is what makes it a reference for what is NOT tumour-specific","notes_pl":"29 rodzaj\u00f3w tkanek od 21 dawc\u00f3w \u2014 tkanki prawid\u0142owe, co czyni go punktem odniesienia dla tego, co NIE jest swoiste dla guza","unit":"class I ligands"},"slug":"hla-ligand-atlas","summary":"A benign-tissue reference peptidome: the set you check a candidate neoantigen against to make sure healthy tissue does not present it too.","tasks":["antigen-presentation"],"url":"https://hla-ligand-atlas.org/","verified_at":"2026-09-05T22:25:59.165582"},{"access":"open","article":"/blog/modele-prezentacji-antygenu","cancer_slugs":["pan-cancer"],"doi":null,"formats":["CSV","PNG"],"hf":null,"huggingface":null,"kind":"dataset","license":"free for academic use","modalities":["immunopeptidomics","protein-sequence"],"name":"MHC Motif Atlas","page":"/ai-oncology/datasets/mhc-motif-atlas","provider":"Gfeller lab, University of Lausanne","size":{"items":1000000,"notes_en":"over a million ligands \u2014 but spread across only about 135 class I molecules, against 30,894 named class I alleles in IPD-IMGT/HLA (June 2026)","notes_pl":"ponad milion ligand\u00f3w \u2014 ale roz\u0142o\u017conych na zaledwie oko\u0142o 135 cz\u0105steczek klasy I, wobec 30 894 nazwanych alleli klasy I w IPD-IMGT/HLA (czerwiec 2026)","unit":"ligands"},"slug":"mhc-motif-atlas","summary":"The reference collection of HLA binding motifs \u2014 and the clearest picture of the field's long tail: a million measured ligands still describe barely 135 of thirty thousand alleles.","tasks":["antigen-presentation"],"url":"http://mhcmotifatlas.org/","verified_at":"2026-09-05T22:25:59.213233"},{"access":"open","article":"/blog/modele-prezentacji-antygenu","cancer_slugs":["pan-cancer"],"doi":null,"formats":["text","CSV"],"hf":null,"huggingface":null,"kind":"registry","license":"CC BY-ND 4.0 (see site terms)","modalities":["genomics","protein-sequence"],"name":"IPD-IMGT/HLA Database","page":"/ai-oncology/datasets/ipd-imgt-hla","provider":"EMBL-EBI / Anthony Nolan Research Institute","size":{"items":30894,"notes_en":"as of June 2026: 9,279 HLA-A, 11,258 HLA-B, 9,416 HLA-C \u2014 while one patient carries at most six","notes_pl":"stan na czerwiec 2026: 9279 HLA-A, 11 258 HLA-B, 9416 HLA-C \u2014 a jeden pacjent ma najwy\u017cej sze\u015b\u0107","unit":"named class I alleles"},"slug":"ipd-imgt-hla","summary":"The naming authority for HLA alleles: the catalogue whose size \u2014 thirty thousand names against roughly a hundred well-measured alleles \u2014 defines the central problem of this field.","tasks":["antigen-presentation"],"url":"https://hla.alleles.org/","verified_at":"2026-09-05T22:25:59.303091"},{"access":"open","article":"/blog/modele-prezentacji-antygenu","cancer_slugs":["pan-cancer"],"doi":null,"formats":["CSV"],"hf":null,"huggingface":null,"kind":"benchmark","license":"public","modalities":["protein-sequence"],"name":"BD2013 \u2014 MHC binding affinity benchmark","page":"/ai-oncology/datasets/bd2013","provider":"IEDB / Kim et al.","size":{"items":176161,"notes_en":"114 alleles across six species \u2014 the historical training core for binding predictors, and a reminder of how small the measured world is","notes_pl":"114 alleli w sze\u015bciu gatunkach \u2014 historyczny rdze\u0144 treningowy modeli wi\u0105zania i przypomnienie, jak ma\u0142y jest \u015bwiat zmierzony","unit":"affinity measurements"},"slug":"bd2013","summary":"The reference affinity dataset behind fifteen years of binding predictors; its size and composition are why dataset composition, not architecture, drives reported performance.","tasks":["peptide-mhc-binding"],"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC4111843/","verified_at":"2026-09-05T22:25:59.496418"}]}
