{"count":2,"items":[{"access":"open","article":null,"cancer_slugs":["pan-cancer"],"doi":null,"formats":["text","JSON","PNG"],"hf":null,"huggingface":null,"kind":"corpus","license":"PubMed metadata public domain; PMC OA articles under CC licences per article","modalities":["literature"],"name":"PubMed / PMC Open Access Subset","page":"/ai-oncology/datasets/pubmed-pmc-oa","provider":"US National Library of Medicine","size":{"items":36000000,"notes_en":"abstracts for all of PubMed; full text and figures for the open-access subset","notes_pl":"abstrakty ca\u0142ego PubMed; pe\u0142ne teksty i ryciny dla podzbioru open-access","unit":"citations (PubMed); millions of full texts in PMC OA"},"slug":"pubmed-pmc-oa","summary":"The literature corpus behind BiomedBERT, BiomedCLIP and CONCH's caption data; the entry point for any oncology NLP or vision-language pretraining.","tasks":["information-extraction","question-answering","image-text-retrieval"],"url":"https://pmc.ncbi.nlm.nih.gov/tools/openftlist/","verified_at":"2026-09-05T22:25:59.060602"},{"access":"open","article":null,"cancer_slugs":["pan-cancer"],"doi":null,"formats":["JSON"],"hf":null,"huggingface":null,"kind":"benchmark","license":"MIT","modalities":["clinical-text"],"name":"MedQA (USMLE)","page":"/ai-oncology/datasets/medqa","provider":"Jin et al. (Columbia University)","size":{"items":12723,"notes_en":"multiple-choice medical licensing exam questions; also Mandarin and Traditional Chinese subsets","notes_pl":"pytania wielokrotnego wyboru z egzamin\u00f3w lekarskich; tak\u017ce podzbiory w mandary\u0144skim i tradycyjnym chi\u0144skim","unit":"questions (English)"},"slug":"medqa","summary":"The exam-style benchmark on which Med-PaLM 2, GPT-4 and MedGemma report headline accuracy; useful for comparing models, not for judging clinical safety.","tasks":["question-answering"],"url":"https://github.com/jind11/MedQA","verified_at":"2026-09-05T22:25:59.099060"}]}
