{"count":26,"items":[{"access":"open","article":null,"cancer_slugs":["invasive-breast-carcinoma"],"doi":null,"formats":["TIFF","CSV"],"hf":null,"huggingface":null,"kind":"benchmark","license":"CC0","modalities":["histopathology"],"name":"CAMELYON16 / CAMELYON17","page":"/ai-oncology/datasets/camelyon16","provider":"Radboud University Medical Center and partners (grand-challenge.org)","size":{"items":1399,"notes_en":"CAMELYON16: 399 slides (270 train / 129 test) with pixel-level metastasis annotations; CAMELYON17: 1,000 slides from 5 centres with patient-level pN stage","notes_pl":"CAMELYON16: 399 preparat\u00f3w (270 tren. / 129 test.) z adnotacjami przerzut\u00f3w na poziomie pikseli; CAMELYON17: 1000 preparat\u00f3w z 5 o\u015brodk\u00f3w ze stadium pN per pacjent","unit":"WSIs"},"slug":"camelyon16","summary":"The canonical whole-slide benchmark for breast cancer lymph-node metastasis detection; still the first sanity check for any new pathology encoder.","tasks":["detection","classification","segmentation"],"url":"https://camelyon17.grand-challenge.org/","verified_at":"2026-09-05T22:25:58.842206"},{"access":"open","article":null,"cancer_slugs":["invasive-breast-carcinoma"],"doi":"10.7937/K9/TCIA.2016.7O02S9CY","formats":["DICOM","CSV"],"hf":null,"huggingface":null,"kind":"dataset","license":"CC BY 3.0","modalities":["mammography"],"name":"CBIS-DDSM \u2014 Curated Breast Imaging Subset of DDSM","page":"/ai-oncology/datasets/cbis-ddsm","provider":"TCIA","size":{"items":2620,"notes_en":"digitised film mammograms with verified pathology","notes_pl":"zdigitalizowane mammografie filmowe z potwierdzon\u0105 patomorfologi\u0105","patients":1566,"unit":"mammography studies"},"slug":"cbis-ddsm","summary":"Most-used open mammography set with pathology-confirmed labels; film-based, so domain shift to modern digital mammography must be handled.","tasks":["detection","classification"],"url":"https://www.cancerimagingarchive.net/collection/cbis-ddsm/","verified_at":"2026-09-05T22:25:58.952228"},{"access":"open","article":null,"cancer_slugs":["pan-cancer"],"doi":null,"formats":["text","JSON","PNG"],"hf":null,"huggingface":null,"kind":"corpus","license":"PubMed metadata public domain; PMC OA articles under CC licences per article","modalities":["literature"],"name":"PubMed / PMC Open Access Subset","page":"/ai-oncology/datasets/pubmed-pmc-oa","provider":"US National Library of Medicine","size":{"items":36000000,"notes_en":"abstracts for all of PubMed; full text and figures for the open-access subset","notes_pl":"abstrakty ca\u0142ego PubMed; pe\u0142ne teksty i ryciny dla podzbioru open-access","unit":"citations (PubMed); millions of full texts in PMC OA"},"slug":"pubmed-pmc-oa","summary":"The literature corpus behind BiomedBERT, BiomedCLIP and CONCH's caption data; the entry point for any oncology NLP or vision-language pretraining.","tasks":["information-extraction","question-answering","image-text-retrieval"],"url":"https://pmc.ncbi.nlm.nih.gov/tools/openftlist/","verified_at":"2026-09-05T22:25:59.060602"},{"access":"open","article":null,"cancer_slugs":["pan-cancer"],"doi":null,"formats":["JSON","CSV"],"hf":null,"huggingface":null,"kind":"registry","license":"public domain","modalities":["clinical-text"],"name":"ClinicalTrials.gov","page":"/ai-oncology/datasets/clinicaltrials-gov","provider":"US National Library of Medicine","size":{"items":500000,"notes_en":"structured eligibility criteria, arms, outcomes and results; full API","notes_pl":"ustrukturyzowane kryteria kwalifikacji, ramiona, punkty ko\u0144cowe i wyniki; pe\u0142ne API","unit":"registered studies"},"slug":"clinicaltrials-gov","summary":"The registry that trial-matching systems such as TrialGPT retrieve from; oncology is its largest therapeutic area.","tasks":["clinical-trial-matching","information-extraction"],"url":"https://clinicaltrials.gov/","verified_at":"2026-09-05T22:25:59.082279"},{"access":"open","article":null,"cancer_slugs":["pan-cancer"],"doi":null,"formats":["JSON"],"hf":null,"huggingface":null,"kind":"benchmark","license":"MIT","modalities":["clinical-text"],"name":"MedQA (USMLE)","page":"/ai-oncology/datasets/medqa","provider":"Jin et al. (Columbia University)","size":{"items":12723,"notes_en":"multiple-choice medical licensing exam questions; also Mandarin and Traditional Chinese subsets","notes_pl":"pytania wielokrotnego wyboru z egzamin\u00f3w lekarskich; tak\u017ce podzbiory w mandary\u0144skim i tradycyjnym chi\u0144skim","unit":"questions (English)"},"slug":"medqa","summary":"The exam-style benchmark on which Med-PaLM 2, GPT-4 and MedGemma report headline accuracy; useful for comparing models, not for judging clinical safety.","tasks":["question-answering"],"url":"https://github.com/jind11/MedQA","verified_at":"2026-09-05T22:25:59.099060"},{"access":"registration","article":null,"cancer_slugs":["pan-cancer"],"doi":null,"formats":["SVS","MAF","VCF","BAM","TSV","JSON"],"hf":null,"huggingface":null,"kind":"registry","license":"NIH GDS Policy \u2014 open tier; controlled tier via dbGaP","modalities":["genomics","transcriptomics","histopathology","radiology-ct","radiology-mri"],"name":"TCGA \u2014 The Cancer Genome Atlas (via NCI Genomic Data Commons)","page":"/ai-oncology/datasets/tcga","provider":"NCI / NHGRI; hosted by the Genomic Data Commons","size":{"items":30000,"notes_en":"33 cancer types; ~11,000 patients with molecular data; diagnostic slides for most cases; matched clinical follow-up","notes_pl":"33 typy nowotwor\u00f3w; ok. 11 000 pacjent\u00f3w z danymi molekularnymi; preparaty diagnostyczne dla wi\u0119kszo\u015bci przypadk\u00f3w; dopasowana obserwacja kliniczna","patients":11000,"unit":"diagnostic + tissue WSIs"},"slug":"tcga","summary":"The reference multi-omics cancer cohort: molecular profiles, clinical outcomes and whole-slide images for 33 cancer types, downloadable through the GDC portal and API.","tasks":["classification","prognosis","survival-analysis","variant-effect","gene-expression"],"url":"https://portal.gdc.cancer.gov/","verified_at":"2026-09-05T22:25:58.814252"},{"access":"registration","article":null,"cancer_slugs":["pan-cancer"],"doi":null,"formats":["TSV","SVS","DICOM","MAF"],"hf":null,"huggingface":null,"kind":"registry","license":"open (proteomics via PDC) / controlled (raw sequencing via dbGaP)","modalities":["proteomics","genomics","transcriptomics","histopathology","radiology-ct"],"name":"CPTAC \u2014 Clinical Proteomic Tumor Analysis Consortium","page":"/ai-oncology/datasets/cptac","provider":"NCI Office of Cancer Clinical Proteomics Research","size":{"notes_en":"10+ tumour types with proteogenomic profiling (proteome, phosphoproteome) on genomically characterised cases; images in TCIA","notes_pl":"10+ typ\u00f3w guz\u00f3w z profilowaniem proteogenomicznym (proteom, fosfoproteom) na przypadkach scharakteryzowanych genomowo; obrazy w TCIA","patients":1000},"slug":"cptac","summary":"Proteogenomic companion to TCGA: the largest public set where protein-level measurements, genomics and images exist for the same tumours.","tasks":["classification","prognosis","gene-expression"],"url":"https://proteomics.cancer.gov/programs/cptac","verified_at":"2026-09-05T22:25:58.828625"},{"access":"open","article":null,"cancer_slugs":["pan-cancer"],"doi":null,"formats":["DICOM","NIfTI","CSV"],"hf":null,"huggingface":null,"kind":"registry","license":"mostly CC BY 3.0/4.0 per collection; some restricted collections","modalities":["radiology-ct","radiology-mri","pet","mammography","histopathology","radiology-xray"],"name":"TCIA \u2014 The Cancer Imaging Archive","page":"/ai-oncology/datasets/tcia","provider":"NCI Cancer Imaging Program; hosted by the University of Arkansas for Medical Sciences","size":{"items":200,"notes_en":"hundreds of collections, tens of thousands of patients; DICOM with linked clinical and sometimes genomic data","notes_pl":"setki kolekcji, dziesi\u0105tki tysi\u0119cy pacjent\u00f3w; DICOM z powi\u0105zanymi danymi klinicznymi, czasem genomowymi","unit":"collections"},"slug":"tcia","summary":"The main public archive of de-identified cancer imaging, organised into collections by disease and modality; the source of most public radiology training data in oncology.","tasks":["segmentation","detection","classification","prognosis"],"url":"https://www.cancerimagingarchive.net/","verified_at":"2026-09-05T22:25:58.836069"},{"access":"registration","article":null,"cancer_slugs":["prostate"],"doi":null,"formats":["TIFF","CSV"],"hf":null,"huggingface":null,"kind":"benchmark","license":"CC BY-NC-SA 4.0 (Kaggle competition rules)","modalities":["histopathology"],"name":"PANDA \u2014 Prostate cANcer graDe Assessment","page":"/ai-oncology/datasets/panda","provider":"Radboud UMC and Karolinska Institutet (Kaggle challenge)","size":{"items":10616,"notes_en":"the largest public prostate biopsy set; Gleason / ISUP grade per biopsy from two centres","notes_pl":"najwi\u0119kszy publiczny zbi\u00f3r biopsji prostaty; stopie\u0144 Gleasona / ISUP per biopsja z dw\u00f3ch o\u015brodk\u00f3w","unit":"biopsy WSIs"},"slug":"panda","summary":"Reference dataset and challenge for AI Gleason grading; the follow-up Nature Medicine paper validated algorithms on external cohorts.","tasks":["classification"],"url":"https://www.kaggle.com/competitions/prostate-cancer-grade-assessment","verified_at":"2026-09-05T22:25:58.872361"},{"access":"registration","article":null,"cancer_slugs":["adult-type-diffuse-gliomas"],"doi":null,"formats":["NIfTI"],"hf":null,"huggingface":null,"kind":"benchmark","license":"challenge terms (registration on Synapse)","modalities":["radiology-mri"],"name":"BraTS \u2014 Brain Tumor Segmentation Challenge","page":"/ai-oncology/datasets/brats","provider":"BraTS organisers (University of Pennsylvania, Indiana University, MICCAI)","size":{"notes_en":"\u22651,250 glioma cases since 2021; later editions add paediatric, metastasis, meningioma and Sub-Saharan Africa tracks","notes_pl":"\u22651250 przypadk\u00f3w glejak\u00f3w od 2021; kolejne edycje dodaj\u0105 \u015bcie\u017cki pediatryczne, przerzut\u00f3w, oponiak\u00f3w i Afryki Subsaharyjskiej","patients":1250,"unit":"multi-parametric MRI cases"},"slug":"brats","summary":"The long-running benchmark for brain tumour segmentation from MRI; nnU-Net-based methods have dominated its leaderboards.","tasks":["segmentation"],"url":"https://www.synapse.org/brats","verified_at":"2026-09-05T22:25:58.886093"},{"access":"open","article":null,"cancer_slugs":["lung-bronchus"],"doi":"10.7937/K9/TCIA.2015.LO9QL9SX","formats":["DICOM","JSON"],"hf":null,"huggingface":null,"kind":"dataset","license":"CC BY 3.0","modalities":["radiology-ct"],"name":"LIDC-IDRI \u2014 Lung Image Database Consortium","page":"/ai-oncology/datasets/lidc-idri","provider":"NCI / TCIA","size":{"items":1018,"notes_en":"nodule annotations by four thoracic radiologists in a two-phase reading","notes_pl":"adnotacje guzk\u00f3w przez czterech radiolog\u00f3w w dwufazowym odczycie","patients":1010,"unit":"CT scans"},"slug":"lidc-idri","summary":"The standard open CT dataset for lung nodule detection and characterisation; basis of the LUNA16 challenge.","tasks":["detection","segmentation","classification"],"url":"https://www.cancerimagingarchive.net/collection/lidc-idri/","verified_at":"2026-09-05T22:25:58.902160"},{"access":"controlled","article":null,"cancer_slugs":["lung-bronchus"],"doi":null,"formats":["DICOM","CSV"],"hf":null,"huggingface":null,"kind":"dataset","license":"NCI CDAS data-use agreement","modalities":["radiology-ct","radiology-xray"],"name":"NLST \u2014 National Lung Screening Trial","page":"/ai-oncology/datasets/nlst","provider":"NCI (Cancer Data Access System)","size":{"notes_en":"randomised trial of low-dose CT vs chest X-ray screening (2002\u20132009) with cancer and mortality follow-up; imaging available for a subset","notes_pl":"randomizowane badanie skriningu niskodawkowym TK vs RTG (2002\u20132009) z obserwacj\u0105 zachorowa\u0144 i zgon\u00f3w; obrazy dost\u0119pne dla podzbioru","patients":53454},"slug":"nlst","summary":"The trial that established LDCT screening; its images plus outcomes trained Sybil and remain the only large public-by-application longitudinal LDCT cohort.","tasks":["risk-prediction","screening","detection"],"url":"https://cdas.cancer.gov/nlst/","verified_at":"2026-09-05T22:25:58.927519"},{"access":"open","article":null,"cancer_slugs":["malignant-melanoma","non-melanoma-skin-cancer-nmsc"],"doi":null,"formats":["JPEG","CSV"],"hf":null,"huggingface":null,"kind":"registry","license":"CC-0 / CC BY-NC per contributor","modalities":["dermoscopy"],"name":"ISIC Archive \u2014 International Skin Imaging Collaboration","page":"/ai-oncology/datasets/isic-archive","provider":"ISIC (Memorial Sloan Kettering and partners)","size":{"items":70000,"notes_en":"tens of thousands of images with diagnosis; annual challenge subsets (e.g. 2020: 33,126 images)","notes_pl":"dziesi\u0105tki tysi\u0119cy obraz\u00f3w z rozpoznaniem; coroczne podzbiory konkursowe (np. 2020: 33 126 obraz\u00f3w)","unit":"dermoscopic images"},"slug":"isic-archive","summary":"The public backbone of skin-cancer AI; strongly skewed towards light skin tones, which every model card built on it should say.","tasks":["classification","detection","segmentation"],"url":"https://www.isic-archive.com/","verified_at":"2026-09-05T22:25:58.972520"},{"access":"registration","article":null,"cancer_slugs":["pan-cancer"],"doi":null,"formats":["MAF","TSV"],"hf":null,"huggingface":null,"kind":"registry","license":"AACR GENIE data-use terms (Synapse registration)","modalities":["genomics","ehr"],"name":"AACR Project GENIE","page":"/ai-oncology/datasets/aacr-genie","provider":"American Association for Cancer Research; hosted on Synapse / cBioPortal","size":{"notes_en":"clinical-grade tumour sequencing from 19+ institutions with limited clinical data; releases twice a year","notes_pl":"sekwencjonowanie guz\u00f3w klasy klinicznej z 19+ instytucji z ograniczonymi danymi klinicznymi; wydania dwa razy w roku","patients":200000},"slug":"aacr-genie","summary":"The largest real-world clinical sequencing registry in oncology \u2014 the place to test whether a variant-effect or biomarker model generalises beyond TCGA.","tasks":["variant-effect","prognosis","survival-analysis"],"url":"https://www.aacr.org/professionals/research/aacr-project-genie/","verified_at":"2026-09-05T22:25:58.995187"},{"access":"open","article":null,"cancer_slugs":["pan-cancer"],"doi":null,"formats":["CSV","Parquet"],"hf":null,"huggingface":null,"kind":"dataset","license":"CC BY 4.0","modalities":["genomics","transcriptomics","molecules"],"name":"DepMap \u2014 Cancer Dependency Map","page":"/ai-oncology/datasets/depmap","provider":"Broad Institute","size":{"items":1900,"notes_en":"genome-wide CRISPR knockout screens, drug sensitivity (PRISM) and multi-omics per line; quarterly releases","notes_pl":"genomowe screeny CRISPR, wra\u017cliwo\u015b\u0107 na leki (PRISM) i multi-omika per linia; wydania kwartalne","unit":"cancer cell lines"},"slug":"depmap","summary":"Public map of cancer vulnerabilities in cell lines \u2014 the training and validation ground for target-discovery and drug-response models.","tasks":["drug-discovery","gene-expression"],"url":"https://depmap.org/portal/","verified_at":"2026-09-05T22:25:59.011882"},{"access":"open","article":null,"cancer_slugs":["pan-cancer"],"doi":null,"formats":["PDB","mmCIF"],"hf":null,"huggingface":null,"kind":"registry","license":"CC0","modalities":["protein-sequence","molecules"],"name":"Protein Data Bank (wwPDB / RCSB)","page":"/ai-oncology/datasets/pdb","provider":"Worldwide Protein Data Bank","size":{"items":220000,"notes_en":"X-ray, cryo-EM and NMR structures of proteins, nucleic acids and complexes; the training ground of AlphaFold","notes_pl":"struktury rentgenowskie, krio-EM i NMR bia\u0142ek, kwas\u00f3w nukleinowych i kompleks\u00f3w; grunt treningowy AlphaFold","unit":"experimental structures"},"slug":"pdb","summary":"The open archive of experimentally solved macromolecular structures, including oncology targets (kinases, KRAS, p53) and their drug complexes.","tasks":["protein-structure","drug-discovery"],"url":"https://www.rcsb.org/","verified_at":"2026-09-05T22:25:59.035915"},{"access":"open","article":"/blog/modele-prezentacji-antygenu","cancer_slugs":["pan-cancer"],"doi":null,"formats":["CSV","JSON"],"hf":null,"huggingface":null,"kind":"registry","license":"free to use; NIAID-funded public resource","modalities":["protein-sequence","immunopeptidomics"],"name":"IEDB \u2014 Immune Epitope Database","page":"/ai-oncology/datasets/iedb","provider":"La Jolla Institute for Immunology, funded by NIAID","size":{"items":1600000,"notes_en":"curated from published literature and direct submissions; includes MHC binding assays, MS-eluted ligands and T-cell assays","notes_pl":"kuratorowane z literatury i zg\u0142osze\u0144 bezpo\u015brednich; zawiera testy wi\u0105zania MHC, ligandy ze spektrometru i testy limfocyt\u00f3w T","unit":"epitope-related records"},"slug":"iedb","summary":"The field's central repository of epitope data and the source of almost every training set for peptide\u2013MHC models \u2014 and of their allele skew.","tasks":["peptide-mhc-binding","antigen-presentation"],"url":"https://www.iedb.org/","verified_at":"2026-09-05T22:25:59.114121"},{"access":"open","article":"/blog/modele-prezentacji-antygenu","cancer_slugs":["pan-cancer"],"doi":null,"formats":["CSV"],"hf":null,"huggingface":null,"kind":"benchmark","license":"public","modalities":["protein-sequence"],"name":"IEDB automated benchmark (MHC class I)","page":"/ai-oncology/datasets/iedb-benchmark","provider":"La Jolla Institute for Immunology","size":{"notes_en":"runs continuously on newly deposited data, before it can leak into anyone's training set","notes_pl":"dzia\u0142a na bie\u017c\u0105co na \u015bwie\u017co deponowanych danych, zanim mog\u0105 trafi\u0107 do czyjegokolwiek zbioru treningowego"},"slug":"iedb-benchmark","summary":"The only prospective, third-party benchmark in the field. Its eight-year summary is sobering: leading methods are statistically indistinguishable, and a new method needs about four years before enough data accumulate to judge it.","tasks":["peptide-mhc-binding","antigen-presentation"],"url":"http://tools.iedb.org/auto_bench/mhci/weekly/","verified_at":"2026-09-05T22:25:59.140589"},{"access":"open","article":"/blog/modele-prezentacji-antygenu","cancer_slugs":["pan-cancer"],"doi":"10.1038/s41587-019-0322-9","formats":["CSV","text"],"hf":null,"huggingface":null,"kind":"dataset","license":"published supplementary data (see papers)","modalities":["immunopeptidomics","protein-sequence"],"name":"Mono-allelic HLA class I peptidome (Sarkizova / Abelin)","page":"/ai-oncology/datasets/monoallelic-peptidome","provider":"Broad Institute / Dana-Farber Cancer Institute","size":{"items":186464,"notes_en":"95 mono-allelic cell lines (31 HLA-A, 40 HLA-B, 21 HLA-C, 3 HLA-G), median 1,860 peptides per allele; the earlier Abelin 2017 set covered 16 alleles and >24,000 peptides","notes_pl":"95 linii monoallelicznych (31 HLA-A, 40 HLA-B, 21 HLA-C, 3 HLA-G), mediana 1860 peptyd\u00f3w na allel; wcze\u015bniejszy zbi\u00f3r Abelin 2017 obj\u0105\u0142 16 alleli i ponad 24 000 peptyd\u00f3w","unit":"peptides"},"slug":"monoallelic-peptidome","summary":"The engineered-cell peptidome that gave the field clean allele labels; fifteen of its alleles had no described motif before, and the panel covers at least one allele in 95% of people worldwide.","tasks":["antigen-presentation","peptide-mhc-binding"],"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC7008090/","verified_at":"2026-09-05T22:25:59.152441"},{"access":"open","article":"/blog/modele-prezentacji-antygenu","cancer_slugs":["pan-cancer"],"doi":null,"formats":["CSV"],"hf":null,"huggingface":null,"kind":"dataset","license":"CC BY 4.0","modalities":["immunopeptidomics"],"name":"HLA Ligand Atlas","page":"/ai-oncology/datasets/hla-ligand-atlas","provider":"University of T\u00fcbingen (Rammensee / Walz groups)","size":{"items":90428,"notes_en":"29 tissue types from 21 donors \u2014 benign tissue, which is what makes it a reference for what is NOT tumour-specific","notes_pl":"29 rodzaj\u00f3w tkanek od 21 dawc\u00f3w \u2014 tkanki prawid\u0142owe, co czyni go punktem odniesienia dla tego, co NIE jest swoiste dla guza","unit":"class I ligands"},"slug":"hla-ligand-atlas","summary":"A benign-tissue reference peptidome: the set you check a candidate neoantigen against to make sure healthy tissue does not present it too.","tasks":["antigen-presentation"],"url":"https://hla-ligand-atlas.org/","verified_at":"2026-09-05T22:25:59.165582"},{"access":"open","article":"/blog/modele-prezentacji-antygenu","cancer_slugs":["pan-cancer"],"doi":null,"formats":["CSV","PNG"],"hf":null,"huggingface":null,"kind":"dataset","license":"free for academic use","modalities":["immunopeptidomics","protein-sequence"],"name":"MHC Motif Atlas","page":"/ai-oncology/datasets/mhc-motif-atlas","provider":"Gfeller lab, University of Lausanne","size":{"items":1000000,"notes_en":"over a million ligands \u2014 but spread across only about 135 class I molecules, against 30,894 named class I alleles in IPD-IMGT/HLA (June 2026)","notes_pl":"ponad milion ligand\u00f3w \u2014 ale roz\u0142o\u017conych na zaledwie oko\u0142o 135 cz\u0105steczek klasy I, wobec 30 894 nazwanych alleli klasy I w IPD-IMGT/HLA (czerwiec 2026)","unit":"ligands"},"slug":"mhc-motif-atlas","summary":"The reference collection of HLA binding motifs \u2014 and the clearest picture of the field's long tail: a million measured ligands still describe barely 135 of thirty thousand alleles.","tasks":["antigen-presentation"],"url":"http://mhcmotifatlas.org/","verified_at":"2026-09-05T22:25:59.213233"},{"access":"open","article":"/blog/modele-prezentacji-antygenu","cancer_slugs":["pan-cancer"],"doi":null,"formats":["text","CSV"],"hf":null,"huggingface":null,"kind":"registry","license":"CC BY-ND 4.0 (see site terms)","modalities":["genomics","protein-sequence"],"name":"IPD-IMGT/HLA Database","page":"/ai-oncology/datasets/ipd-imgt-hla","provider":"EMBL-EBI / Anthony Nolan Research Institute","size":{"items":30894,"notes_en":"as of June 2026: 9,279 HLA-A, 11,258 HLA-B, 9,416 HLA-C \u2014 while one patient carries at most six","notes_pl":"stan na czerwiec 2026: 9279 HLA-A, 11 258 HLA-B, 9416 HLA-C \u2014 a jeden pacjent ma najwy\u017cej sze\u015b\u0107","unit":"named class I alleles"},"slug":"ipd-imgt-hla","summary":"The naming authority for HLA alleles: the catalogue whose size \u2014 thirty thousand names against roughly a hundred well-measured alleles \u2014 defines the central problem of this field.","tasks":["antigen-presentation"],"url":"https://hla.alleles.org/","verified_at":"2026-09-05T22:25:59.303091"},{"access":"open","article":"/blog/modele-prezentacji-antygenu","cancer_slugs":["pan-cancer"],"doi":null,"formats":["CSV"],"hf":null,"huggingface":null,"kind":"benchmark","license":"public","modalities":["protein-sequence"],"name":"BD2013 \u2014 MHC binding affinity benchmark","page":"/ai-oncology/datasets/bd2013","provider":"IEDB / Kim et al.","size":{"items":176161,"notes_en":"114 alleles across six species \u2014 the historical training core for binding predictors, and a reminder of how small the measured world is","notes_pl":"114 alleli w sze\u015bciu gatunkach \u2014 historyczny rdze\u0144 treningowy modeli wi\u0105zania i przypomnienie, jak ma\u0142y jest \u015bwiat zmierzony","unit":"affinity measurements"},"slug":"bd2013","summary":"The reference affinity dataset behind fifteen years of binding predictors; its size and composition are why dataset composition, not architecture, drives reported performance.","tasks":["peptide-mhc-binding"],"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC4111843/","verified_at":"2026-09-05T22:25:59.496418"},{"access":"open","article":null,"cancer_slugs":[],"doi":null,"formats":["other"],"hf":null,"huggingface":null,"kind":"dataset","license":"Apache-2.0","modalities":["transcriptomics","single-cell"],"name":"Genecorpus-30M","page":"/ai-oncology/datasets/genecorpus-30m","provider":"Theodoris/Ellinor lab (Broad Institute, Massachusetts General Hospital)","size":{"cells":"27 406 217 komorek po filtrach jakosci (z 29 900 531 zebranych)","headline":"~30 mln ludzkich transkryptomow pojedynczych komorek","source_datasets":"561 publicznie dostepnych zbiorow danych"},"slug":"genecorpus-30m","summary":"Genecorpus-30M is a pretraining corpus of about 30 million human single-cell transcriptomes assembled from 561 publicly available datasets, of which 27,406,217 cells passed quality filters. It was built to pretrain Geneformer (Theodoris et al., Nature 2023) and is released under Apache-2.0. Crucially for oncology use: cells with high mutational burden \u2014 malignant cells and immortalized cell lines \u2014 were deliberately excluded from the corpus.","tasks":["gene-expression","single-cell"],"url":"https://huggingface.co/datasets/ctheodoris/Genecorpus-30M","verified_at":"2026-09-08T04:55:51.018757"},{"access":"open","article":null,"cancer_slugs":["invasive-breast-carcinoma"],"doi":null,"formats":["h5ad","PNG","Parquet"],"hf":{"downloads":3508,"fetched_at":"2026-09-09T21:33:09Z","last_modified":"2024-05-25","license":"cc0-1.0","likes":12},"huggingface":"https://huggingface.co/datasets/1aurent/PatchCamelyon","kind":"benchmark","license":"CC0","modalities":["histopathology"],"name":"PatchCamelyon (PCam)","page":"/ai-oncology/datasets/patchcamelyon","provider":"Veeling et al.; mirrored on Hugging Face","size":{"items":327680,"notes_en":"derived from CAMELYON16; binary label = tumour tissue in the central 32\u00d732 region","notes_pl":"pochodna CAMELYON16; etykieta binarna = tkanka guza w centralnym obszarze 32\u00d732","unit":"96\u00d796 patches"},"slug":"patchcamelyon","summary":"Small, fast, fully open patch-classification benchmark \u2014 the standard smoke test for image encoders in pathology and a common zero-shot evaluation set.","tasks":["classification"],"url":"https://github.com/basveeling/pcam","verified_at":"2026-09-05T22:25:58.858045"},{"access":"open","article":null,"cancer_slugs":[],"doi":null,"formats":["FASTQ","BAM","VCF"],"hf":null,"huggingface":null,"kind":null,"license":"Materia\u0142 referencyjny NIST; zgoda dawc\u00f3w z PGP na redystrybucj\u0119 komercyjn\u0105","modalities":["genomics"],"name":"Genome in a Bottle (GIAB)","page":"/ai-oncology/datasets/giab-genome-in-a-bottle","provider":"National Institute of Standards and Technology (NIST), USA","size":{"items":7,"notes_en":"Seven characterised genomes: the pilot genome NA12878/HG001 from the HapMap collection and two Personal Genome Project trios \u2014 Ashkenazi Jewish (HG002, HG003, HG004) and Han Chinese (HG005, HG006, HG007).","notes_pl":"Siedem scharakteryzowanych genom\u00f3w: genom pilota\u017cowy NA12878/HG001 z kolekcji HapMap oraz dwa trio z Personal Genome Project \u2014 aszkenazyjskie (HG002, HG003, HG004) i chi\u0144skie Han (HG005, HG006, HG007).","patients":7,"unit":"genomy referencyjne"},"slug":"giab-genome-in-a-bottle","summary":"A public benchmark resource run by NIST within a consortium of public bodies, companies and academic centres. It contains seven thoroughly characterised human genomes with benchmark variant sets and high-confidence regions, and is used to measure how accurately a bioinformatics pipeline calls variants. It is not an oncology dataset and not an imaging dataset \u2014 it is a reference point for genome readout itself. The consortium works with the GA4GH Benchmarking Team on comparison standards. Distribution: the NIST reference material shop, the Coriell Institute and the Personal Genome Project.","tasks":[],"url":"https://www.nist.gov/programs-projects/genome-bottle","verified_at":null}]}
