{"count":1,"items":[{"access":"open","article":null,"cancer_slugs":[],"doi":null,"formats":["other"],"hf":null,"huggingface":null,"kind":"dataset","license":"Apache-2.0","modalities":["transcriptomics","single-cell"],"name":"Genecorpus-30M","page":"/ai-oncology/datasets/genecorpus-30m","provider":"Theodoris/Ellinor lab (Broad Institute, Massachusetts General Hospital)","size":{"cells":"27 406 217 komorek po filtrach jakosci (z 29 900 531 zebranych)","headline":"~30 mln ludzkich transkryptomow pojedynczych komorek","source_datasets":"561 publicznie dostepnych zbiorow danych"},"slug":"genecorpus-30m","summary":"Genecorpus-30M is a pretraining corpus of about 30 million human single-cell transcriptomes assembled from 561 publicly available datasets, of which 27,406,217 cells passed quality filters. It was built to pretrain Geneformer (Theodoris et al., Nature 2023) and is released under Apache-2.0. Crucially for oncology use: cells with high mutational burden \u2014 malignant cells and immortalized cell lines \u2014 were deliberately excluded from the corpus.","tasks":["gene-expression","single-cell"],"url":"https://huggingface.co/datasets/ctheodoris/Genecorpus-30M","verified_at":"2026-09-08T04:55:51.018757"}]}
