Files

5980 lines
147 KiB
JSON
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
---
title: "Resources"
task: ""
lineage_type: import
upstream_source: https://github.com/inoue0426/awesome-computational-biology/blob/7a064bf0/data/resources.json
upstream_sha: 7a064bf0
imported_at: 2026-08-08
prompt_class: catalogue
upstream_changes: accepted
author: upstream
validated: false
---
[
{
"id": "chembl_web_services",
"name": "ChEMBL Web Services",
"type": "api",
"url": "https://www.ebi.ac.uk/chembl/ws",
"description": "REST API for bioactive molecules, targets, and bioassays.",
"tags": [
"api"
],
"tasks": [],
"modalities": [
"chemical-structure"
],
"organism": [],
"api": true,
"entities": [
"molecule",
"protein"
],
"documentation": "https://www.ebi.ac.uk/chembl/api/data/docs",
"last_checked": "2026-08-08",
"metadata_sources": [
"https://www.ebi.ac.uk/chembl/api/data/docs"
]
},
{
"id": "clinicaltrials_gov_api",
"name": "ClinicalTrials.gov API",
"type": "api",
"url": "https://clinicaltrials.gov/api/gui",
"description": "API for querying clinical trial metadata and results.",
"tags": [
"api"
],
"tasks": [],
"modalities": [
"clinical"
],
"organism": [],
"api": true,
"entities": [
"disease",
"drug"
],
"documentation": "https://clinicaltrials.gov/data-api/api",
"last_checked": "2026-08-08",
"metadata_sources": [
"https://clinicaltrials.gov/data-api/api"
]
},
{
"id": "ensembl_rest_api",
"name": "Ensembl REST API",
"type": "api",
"url": "https://rest.ensembl.org/",
"description": "API for genomic annotations, variants, genes, and comparative genomics.",
"tags": [
"api"
],
"tasks": [],
"modalities": [
"genomics"
],
"organism": [],
"api": true,
"entities": [
"gene",
"genome",
"transcript",
"variant"
],
"documentation": "https://rest.ensembl.org/",
"last_checked": "2026-08-08",
"metadata_sources": [
"https://rest.ensembl.org/"
]
},
{
"id": "kegg_rest_api",
"name": "KEGG REST API",
"type": "api",
"url": "https://www.kegg.jp/kegg/rest/keggapi.html",
"description": "API for accessing KEGG pathways, compounds, genes, and reactions.",
"tags": [
"api"
],
"tasks": [],
"modalities": [],
"organism": [],
"api": true,
"entities": [
"compound",
"gene",
"pathway"
],
"documentation": "https://www.kegg.jp/kegg/rest/keggapi.html",
"last_checked": "2026-08-08",
"metadata_sources": [
"https://www.kegg.jp/kegg/rest/keggapi.html"
]
},
{
"id": "ncbi_e_utilities",
"name": "NCBI E-utilities",
"type": "api",
"url": "https://www.ncbi.nlm.nih.gov/books/NBK25501/",
"description": "Unified APIs for accessing NCBI databases (Gene, GEO, SRA, PubChem, etc).",
"tags": [
"api"
],
"tasks": [],
"modalities": [
"genomics",
"transcriptomics"
],
"organism": [],
"api": true,
"entities": [
"gene",
"genome",
"protein",
"transcript",
"variant"
],
"documentation": "https://www.ncbi.nlm.nih.gov/books/NBK25501/",
"last_checked": "2026-08-08",
"metadata_sources": [
"https://www.ncbi.nlm.nih.gov/books/NBK25501/"
]
},
{
"id": "open_targets_platform_api",
"name": "Open Targets Platform API",
"type": "api",
"url": "https://platform.opentargets.org/api",
"description": "API for targetโ€“disease associations integrating genetics, genomics, and drug data.",
"tags": [
"api"
],
"tasks": [],
"modalities": [
"genomics",
"knowledge-graph"
],
"organism": [],
"api": true,
"entities": [
"disease",
"drug",
"gene",
"variant"
],
"documentation": "https://platform.opentargets.org/api",
"last_checked": "2026-08-08",
"metadata_sources": [
"https://platform.opentargets.org/api"
]
},
{
"id": "pubmed_e_utilities_esearch_efetch",
"name": "PubMed E-utilities (esearch/efetch)",
"type": "api",
"url": "https://www.nlm.nih.gov/dataguide/edirect/esearch.html",
"description": "APIs for searching and retrieving biomedical literature from PubMed.",
"tags": [
"api"
],
"tasks": [],
"modalities": [],
"organism": [],
"api": true,
"documentation": "https://www.ncbi.nlm.nih.gov/books/NBK25501/",
"last_checked": "2026-08-08",
"metadata_sources": [
"https://www.ncbi.nlm.nih.gov/books/NBK25501/"
]
},
{
"id": "uniprot_rest_api",
"name": "UniProt REST API",
"type": "api",
"url": "https://www.uniprot.org/help/api",
"description": "Programmatic access to protein sequence and functional annotation data.",
"tags": [
"api"
],
"tasks": [],
"modalities": [
"protein-sequence",
"proteomics"
],
"organism": [],
"api": true,
"entities": [
"protein"
],
"documentation": "https://www.uniprot.org/help/api",
"last_checked": "2026-08-08",
"metadata_sources": [
"https://www.uniprot.org/help/api"
]
},
{
"id": "1000_genomes_project",
"name": "1000 Genomes Project",
"type": "benchmark",
"url": "https://www.internationalgenome.org/",
"description": "Reference panel of human genetic variation from 2,504 individuals across 26 populations.",
"tags": [
"benchmarks-and-datasets"
],
"tasks": [],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "bace",
"name": "BACE",
"type": "benchmark",
"url": "https://www.kaggle.com/datasets/gokturkkoch/bace",
"description": "Binary classification and regression dataset for ฮฒ-secretase 1 (BACE-1) inhibitor binding affinity.",
"tags": [
"benchmarks-and-datasets"
],
"tasks": [
"classification",
"regression"
],
"modalities": [
"chemical-structure"
],
"organism": [],
"api": false,
"entities": [
"molecule",
"protein"
],
"last_checked": "2026-08-08",
"metadata_sources": [
"https://www.kaggle.com/datasets/gokturkkoch/bace"
]
},
{
"id": "beat_aml",
"name": "BEAT AML",
"type": "benchmark",
"url": "https://biodev.github.io/BeatAML2/",
"description": "Functional ex vivo drug sensitivity measurements paired with genomics for acute myeloid leukemia.",
"tags": [
"benchmarks-and-datasets"
],
"tasks": [
"drug-response-prediction"
],
"modalities": [
"genomics"
],
"organism": [],
"api": false,
"entities": [
"cell",
"disease",
"drug",
"gene"
],
"last_checked": "2026-08-08",
"metadata_sources": [
"https://biodev.github.io/BeatAML2/"
]
},
{
"id": "bento",
"name": "Bento",
"type": "benchmark",
"url": "https://github.com/LigandPro/Bento",
"description": "Protein-ligand docking benchmark covering rigid, flexible, de novo, blind, induced-fit, and covalent docking tasks.",
"tags": [
"benchmarks-and-datasets"
],
"tasks": [],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "bindingdb_curated_sets",
"name": "BindingDB Curated Sets",
"type": "benchmark",
"url": "https://www.bindingdb.org/rwd/bind/chemsearch/marvin/SDFdownload.jsp?all_download=yes",
"description": "Curated binding affinity datasets for proteinโ€“ligand interaction benchmarking.",
"tags": [
"benchmarks-and-datasets"
],
"tasks": [
"drug-target-interaction"
],
"modalities": [
"chemical-structure"
],
"organism": [],
"api": false,
"entities": [
"molecule",
"protein"
],
"last_checked": "2026-08-08",
"metadata_sources": [
"https://www.bindingdb.org/"
]
},
{
"id": "cancer_therapeutics_response_portal_ctrp",
"name": "Cancer Therapeutics Response Portal (CTRP)",
"type": "benchmark",
"url": "https://portals.broadinstitute.org/ctrp/",
"description": "Drug sensitivity profiles across ~900 cancer cell lines for >400 compounds.",
"tags": [
"benchmarks-and-datasets"
],
"tasks": [
"drug-response-prediction"
],
"modalities": [],
"organism": [],
"api": false,
"entities": [
"cell",
"drug"
],
"last_checked": "2026-08-08",
"metadata_sources": [
"https://portals.broadinstitute.org/ctrp/"
]
},
{
"id": "clintox",
"name": "ClinTox",
"type": "benchmark",
"url": "https://tdcommons.ai/single_pred_tasks/tox/#clintox",
"description": "Clinical toxicity dataset contrasting FDA-approved drugs with those that failed clinical trials due to toxicity.",
"tags": [
"benchmarks-and-datasets"
],
"tasks": [
"classification"
],
"modalities": [
"clinical"
],
"organism": [],
"api": false,
"entities": [
"drug"
],
"last_checked": "2026-08-08",
"metadata_sources": [
"https://tdcommons.ai/single_pred_tasks/tox/#clintox"
]
},
{
"id": "cptac_clinical_proteomic_tumor_analysis_consortium",
"name": "CPTAC (Clinical Proteomic Tumor Analysis Consortium)",
"type": "benchmark",
"url": "https://proteomics.cancer.gov/programs/cptac",
"description": "Multi-omic proteogenomic datasets for multiple cancer types linking proteomics with genomics.",
"tags": [
"benchmarks-and-datasets"
],
"tasks": [],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "crossdocked2020",
"name": "CrossDocked2020",
"type": "benchmark",
"url": "https://arxiv.org/abs/2001.01037",
"description": "Large-scale dataset for structure-based virtual screening.",
"tags": [
"benchmarks-and-datasets"
],
"tasks": [],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "dud_e_directory_of_useful_decoys_enhanced",
"name": "DUD-E (Directory of Useful Decoys, Enhanced)",
"type": "benchmark",
"url": "http://dude.docking.org/",
"description": "Structure-based virtual screening benchmark with active ligands and challenging decoy sets across diverse protein targets.",
"tags": [
"benchmarks-and-datasets"
],
"tasks": [],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "flip_fitness_landscape_inference_for_proteins",
"name": "FLIP (Fitness Landscape Inference for Proteins)",
"type": "benchmark",
"url": "https://github.com/J-SNACKKB/FLIP",
"description": "Benchmark collection of protein fitness landscape datasets for evaluating protein ML models.",
"tags": [
"benchmarks-and-datasets"
],
"tasks": [],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "guacamol",
"name": "GuacaMol",
"type": "benchmark",
"url": "https://github.com/BenevolentAI/guacamol",
"description": "Benchmark suite for generative molecular design models.",
"tags": [
"benchmarks-and-datasets"
],
"tasks": [
"molecular-generation"
],
"modalities": [
"chemical-structure"
],
"organism": [],
"api": false,
"entities": [
"molecule"
],
"github": "https://github.com/BenevolentAI/guacamol",
"last_checked": "2026-08-08",
"metadata_sources": [
"https://github.com/BenevolentAI/guacamol"
]
},
{
"id": "hest_xenium_virtual_spatial_transcriptomics",
"name": "HEST Xenium virtual spatial transcriptomics",
"type": "benchmark",
"url": "https://huggingface.co/datasets/ratschlab/HEST_Xenium_virtual_spatial_transcriptomics",
"description": "DeepSpot-M predicted transcriptome-wide ST for 59 HEST-1k 10x Xenium samples (~13.3M cells) (gated). Paper: [DeepSpot-M](https://www.medrxiv.org/content/10.64898/2026.06.19.26356060v1).",
"tags": [
"benchmarks-and-datasets"
],
"tasks": [
"regression"
],
"modalities": [
"histopathology",
"spatial-transcriptomics",
"transcriptomics"
],
"organism": [],
"api": false,
"entities": [
"cell",
"gene",
"tissue"
],
"documentation": "https://huggingface.co/datasets/ratschlab/HEST_Xenium_virtual_spatial_transcriptomics",
"last_checked": "2026-08-08",
"metadata_sources": [
"https://huggingface.co/datasets/ratschlab/HEST_Xenium_virtual_spatial_transcriptomics"
]
},
{
"id": "jump_cell_painting_datasets",
"name": "JUMP Cell Painting Datasets",
"type": "benchmark",
"url": "https://github.com/jump-cellpainting/datasets",
"description": "Consortium-scale cell imaging perturbation datasets (chemical and genetic) for phenotypic profiling and drug discovery research.",
"tags": [
"benchmarks-and-datasets"
],
"tasks": [],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "lincs_l1000",
"name": "LINCS L1000",
"type": "benchmark",
"url": "https://lincsproject.org/LINCS/tools/workflows/find-the-best-place-to-obtain-the-lincs-l1000-data",
"description": "Gene expression profiles (978 landmark genes) for >20,000 chemical and genetic perturbations across cell lines.",
"tags": [
"benchmarks-and-datasets"
],
"tasks": [
"perturbation-prediction"
],
"modalities": [
"transcriptomics"
],
"organism": [],
"api": false,
"entities": [
"cell",
"compound",
"gene"
],
"last_checked": "2026-08-08",
"metadata_sources": [
"https://lincsproject.org/LINCS/tools/workflows/find-the-best-place-to-obtain-the-lincs-l1000-data"
]
},
{
"id": "moleculenet",
"name": "MoleculeNet",
"type": "benchmark",
"url": "http://moleculenet.ai/",
"description": "Benchmark datasets for molecular machine learning.",
"tags": [
"benchmarks-and-datasets"
],
"tasks": [
"classification",
"regression"
],
"modalities": [
"chemical-structure"
],
"organism": [],
"api": false,
"entities": [
"molecule"
],
"github": "https://github.com/deepchem/moleculenet",
"last_checked": "2026-08-08",
"metadata_sources": [
"https://github.com/deepchem/moleculenet"
]
},
{
"id": "moses",
"name": "MOSES",
"type": "benchmark",
"url": "https://github.com/molecularsets/moses",
"description": "Benchmarking platform for molecular generation models.",
"tags": [
"benchmarks-and-datasets"
],
"tasks": [],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "ogb_open_graph_benchmark",
"name": "OGB (Open Graph Benchmark)",
"type": "benchmark",
"url": "https://ogb.stanford.edu/",
"description": "Large-scale graph ML benchmark suite including biological datasets such as ogbl-ppa (protein-protein associations) and ogbg-molhiv.",
"tags": [
"benchmarks-and-datasets"
],
"tasks": [],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "openbiolink",
"name": "OpenBioLink",
"type": "benchmark",
"url": "https://github.com/OpenBioLink/OpenBioLink",
"description": "Benchmark datasets for biological knowledge graph completion.",
"tags": [
"benchmarks-and-datasets"
],
"tasks": [],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "pharmgkb",
"name": "PharmGKB",
"type": "benchmark",
"url": "https://www.pharmgkb.org/",
"description": "Curated pharmacogenomics dataset linking genetic variants to drug response phenotypes across thousands of drugs.",
"tags": [
"benchmarks-and-datasets"
],
"tasks": [
"drug-response-prediction"
],
"modalities": [
"clinical",
"genomics"
],
"organism": [],
"api": false,
"entities": [
"drug",
"gene",
"phenotype",
"variant"
],
"last_checked": "2026-08-08",
"metadata_sources": [
"https://www.pharmgkb.org/"
]
},
{
"id": "pk_db",
"name": "PK-DB",
"type": "benchmark",
"url": "https://pk-db.com/",
"description": "Open database of experimental pharmacokinetics (PK) and ADME data from clinical and preclinical studies.",
"tags": [
"benchmarks-and-datasets"
],
"tasks": [],
"modalities": [
"clinical"
],
"organism": [],
"api": false,
"entities": [
"drug"
],
"last_checked": "2026-08-08",
"metadata_sources": [
"https://pk-db.com/"
]
},
{
"id": "prism",
"name": "PRISM",
"type": "benchmark",
"url": "https://depmap.org/portal/prism/",
"description": "Cancer drug sensitivity profiling of >4,500 drugs across >900 cancer cell lines using pooled-cell-line barcoding.",
"tags": [
"benchmarks-and-datasets"
],
"tasks": [
"drug-response-prediction"
],
"modalities": [],
"organism": [],
"api": false,
"entities": [
"cell",
"drug"
],
"last_checked": "2026-08-08",
"metadata_sources": [
"https://depmap.org/portal/prism/"
]
},
{
"id": "proteingym",
"name": "ProteinGym",
"type": "benchmark",
"url": "https://github.com/OATML-Markslab/ProteinGym",
"description": "Large-scale benchmark of deep mutational scanning assays for evaluating protein fitness landscape models.",
"tags": [
"benchmarks-and-datasets"
],
"tasks": [
"regression"
],
"modalities": [
"protein-sequence"
],
"organism": [],
"api": false,
"entities": [
"protein"
],
"github": "https://github.com/OATML-Markslab/ProteinGym",
"last_checked": "2026-08-08",
"metadata_sources": [
"https://github.com/OATML-Markslab/ProteinGym"
]
},
{
"id": "qm9",
"name": "QM9",
"type": "benchmark",
"url": "https://figshare.com/collections/Quantum_chemistry_structures_and_properties_of_134_kilo_molecules/978904",
"description": "Quantum chemistry properties for 134K stable small organic molecules computed at DFT level.",
"tags": [
"benchmarks-and-datasets"
],
"tasks": [],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "scib_single_cell_integration_benchmarks",
"name": "scIB (Single-cell Integration Benchmarks)",
"type": "benchmark",
"url": "https://github.com/theislab/scib",
"description": "Comprehensive benchmarking framework for single-cell data integration methods.",
"tags": [
"benchmarks-and-datasets"
],
"tasks": [],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "scperturb",
"name": "scPerturb",
"type": "benchmark",
"url": "https://github.com/sanderlab/scPerturb",
"description": "Curated and continuously updated single-cell perturbation data resource spanning CRISPR and drug perturbation studies.",
"tags": [
"benchmarks-and-datasets"
],
"tasks": [
"perturbation-prediction"
],
"modalities": [
"single-cell-rna-seq"
],
"organism": [],
"api": false,
"entities": [
"cell",
"drug",
"gene"
],
"github": "https://github.com/sanderlab/scPerturb",
"last_checked": "2026-08-08",
"metadata_sources": [
"https://github.com/sanderlab/scPerturb"
]
},
{
"id": "sider_side_effect_resource",
"name": "SIDER (Side Effect Resource)",
"type": "benchmark",
"url": "http://sideeffects.embl.de/",
"description": "Database of 1,430 approved drugs with their recorded adverse drug reactions across 27 system-organ classes.",
"tags": [
"benchmarks-and-datasets"
],
"tasks": [],
"modalities": [
"clinical"
],
"organism": [],
"api": false,
"entities": [
"drug",
"phenotype"
],
"last_checked": "2026-08-08",
"metadata_sources": [
"http://sideeffects.embl.de/"
]
},
{
"id": "tabula_muris",
"name": "Tabula Muris",
"type": "benchmark",
"url": "https://tabula-muris.ds.czbiohub.org/",
"description": "Comprehensive single-cell atlas of 20 mouse organs and tissues, enabling cross-tissue and cross-species comparisons.",
"tags": [
"benchmarks-and-datasets"
],
"tasks": [],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "tabula_sapiens",
"name": "Tabula Sapiens",
"type": "benchmark",
"url": "https://tabula-sapiens-portal.ds.czbiohub.org/",
"description": "Comprehensive human single-cell atlas of ~500K cells from 24 organs and tissues across multiple donors.",
"tags": [
"benchmarks-and-datasets"
],
"tasks": [],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "tape_tasks_assessing_protein_embeddings",
"name": "TAPE (Tasks Assessing Protein Embeddings)",
"type": "benchmark",
"url": "https://github.com/songlab-cal/tape",
"description": "Benchmark suite of five biologically meaningful semi-supervised learning tasks for evaluating protein representations.",
"tags": [
"benchmarks-and-datasets"
],
"tasks": [],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "tcga_virtual_spatial_transcriptomics_atlas",
"name": "TCGA virtual spatial transcriptomics atlas",
"type": "benchmark",
"url": "https://huggingface.co/datasets/ratschlab/TCGA_virtual_spatial_transcriptomics_atlas",
"description": "DeepSpot-M predicted transcriptome-wide ST for TCGA H&E (FF + FFPE; 28,664 slides / 32 cancer types; gated). Paper: [DeepSpot-M](https://www.medrxiv.org/content/10.64898/2026.06.19.26356060v1).",
"tags": [
"benchmarks-and-datasets"
],
"tasks": [],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "the_cancer_genome_atlas_tcga",
"name": "The Cancer Genome Atlas (TCGA)",
"type": "benchmark",
"url": "https://www.cancer.gov/about-nci/organization/ccg/research/structural-genomics/tcga",
"description": "Comprehensive multi-omics (genomics, transcriptomics, proteomics, methylation) dataset for 33 cancer types across ~11,000 patients.",
"tags": [
"benchmarks-and-datasets"
],
"tasks": [],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "therapeutics_data_commons_tdc",
"name": "Therapeutics Data Commons (TDC)",
"type": "benchmark",
"url": "https://tdcommons.ai/",
"description": "Unified benchmark suite covering ADMET, drug-target interaction, drug response, and more.",
"tags": [
"benchmarks-and-datasets"
],
"tasks": [],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "tox21",
"name": "Tox21",
"type": "benchmark",
"url": "https://tripod.nih.gov/tox21/challenge/",
"description": "12,707 compounds tested in 12 nuclear receptor and stress-response pathway biochemical assays for toxicity prediction.",
"tags": [
"benchmarks-and-datasets"
],
"tasks": [],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "uk_biobank",
"name": "UK Biobank",
"type": "benchmark",
"url": "https://www.ukbiobank.ac.uk/",
"description": "Large-scale biomedical database of ~500K participants with genetic, imaging, and health data for population genetics and disease studies.",
"tags": [
"benchmarks-and-datasets"
],
"tasks": [],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "10x_genomics_dataset",
"name": "10x Genomics Dataset",
"type": "database",
"url": "https://www.10xgenomics.com/resources/datasets",
"description": "Collection of single-cell datasets.",
"tags": [
"genome"
],
"tasks": [],
"modalities": [
"Genomics"
],
"organism": [],
"api": false
},
{
"id": "alphafold_protein_structure_database",
"name": "AlphaFold Protein Structure Database",
"type": "database",
"url": "https://alphafold.ebi.ac.uk/api-docs",
"description": "3D protein structure predictions.",
"tags": [
"protein"
],
"tasks": [],
"modalities": [
"Protein"
],
"organism": [],
"api": false
},
{
"id": "bindingdb",
"name": "BindingDB",
"type": "database",
"url": "https://www.bindingdb.org/rwd/bind/index.jsp",
"description": "Compounds and target database.",
"tags": [
"chemical-protein-interaction",
"interaction"
],
"tasks": [],
"modalities": [
"Protein",
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "biocyc",
"name": "BioCyc",
"type": "database",
"url": "https://biocyc.org/",
"description": "Collection of pathway/genome databases across thousands of organisms.",
"tags": [
"pathway"
],
"tasks": [],
"modalities": [
"Pathway"
],
"organism": [],
"api": false
},
{
"id": "biogrid",
"name": "BioGRID",
"type": "database",
"url": "https://thebiogrid.org/",
"description": "Protein, genetic, and chemical interactions.",
"tags": [
"interaction",
"protein-protein-interaction"
],
"tasks": [],
"modalities": [
"Protein"
],
"organism": [],
"api": false
},
{
"id": "cancer_cell_line_encyclopedia",
"name": "Cancer Cell Line Encyclopedia",
"type": "database",
"url": "https://sites.broadinstitute.org/ccle/",
"description": "Database of ~1000 cancer cell lines.",
"tags": [
"drug-cell-line-response",
"interaction"
],
"tasks": [],
"modalities": [
"Gene Expression",
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "catalogue_of_somatic_mutations_in_cancer_cosmic",
"name": "Catalogue Of Somatic Mutations In Cancer (COSMIC)",
"type": "database",
"url": "https://cancer.sanger.ac.uk/cosmic",
"description": "Resource on somatic mutations in cancers.",
"tags": [
"genome"
],
"tasks": [],
"modalities": [
"Genomics"
],
"organism": [],
"api": false
},
{
"id": "cath_database",
"name": "CATH database",
"type": "database",
"url": "https://www.cathdb.info/",
"description": "Hierarchical classification of protein domain structures.",
"tags": [
"protein"
],
"tasks": [],
"modalities": [
"Protein"
],
"organism": [],
"api": false
},
{
"id": "cbioportal",
"name": "cBioPortal",
"type": "database",
"url": "https://www.cbioportal.org/",
"description": "Cancer genomics database; aggregating many patient datasets.",
"tags": [
"genome"
],
"tasks": [],
"modalities": [
"Genomics"
],
"organism": [],
"api": false
},
{
"id": "cellminer_cross_database_cellminercdb",
"name": "CellMiner Cross Database (CellMinerCDB)",
"type": "database",
"url": "https://discover.nci.nih.gov/cellminercdb/",
"description": "Integrates multiple cancer cell line databases.",
"tags": [
"drug-cell-line-response",
"interaction"
],
"tasks": [
"drug-response-prediction"
],
"modalities": [
"genomics"
],
"organism": [],
"api": false,
"entities": [
"cell",
"drug",
"gene"
],
"last_checked": "2026-08-08",
"metadata_sources": [
"https://discover.nci.nih.gov/cellminercdb/"
]
},
{
"id": "chebi",
"name": "ChEBI",
"type": "database",
"url": "https://www.ebi.ac.uk/chebi/",
"description": "Database focused on small chemical compounds.",
"tags": [
"compound"
],
"tasks": [],
"modalities": [
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "chembl",
"name": "ChEMBL",
"type": "database",
"url": "https://www.ebi.ac.uk/chembl/",
"description": "Bioactive molecules with drug-like properties.",
"tags": [
"compound"
],
"tasks": [],
"modalities": [
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "chemspider",
"name": "ChemSpider",
"type": "database",
"url": "http://www.chemspider.com/",
"description": "Chemical structure database.",
"tags": [
"compound"
],
"tasks": [],
"modalities": [
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "clinicaltrials_gov",
"name": "ClinicalTrials.gov",
"type": "database",
"url": "https://clinicaltrials.gov/",
"description": "Privately and publicly funded clinical studies.",
"tags": [
"clinical-trial"
],
"tasks": [],
"modalities": [
"Clinical"
],
"organism": [],
"api": false
},
{
"id": "comparative_toxicogenomics_database",
"name": "Comparative Toxicogenomics Database",
"type": "database",
"url": "http://ctdbase.org/",
"description": "Chemical-gene interactions, chemical-disease and gene-disease associations, chemical-phenotype associations.",
"tags": [
"drug-gene-interaction",
"interaction"
],
"tasks": [],
"modalities": [
"Gene",
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "critical_assessment_of_structure_prediction_casp",
"name": "Critical Assessment of Structure Prediction (CASP)",
"type": "database",
"url": "https://predictioncenter.org/",
"description": "Assessing methods for protein structure prediction.",
"tags": [
"protein"
],
"tasks": [],
"modalities": [
"Protein"
],
"organism": [],
"api": false
},
{
"id": "cz_cellxgene",
"name": "CZ CELLxGENE",
"type": "database",
"url": "https://cellxgene.cziscience.com/",
"description": "Single-cell dataset repository and interactive explorer from the Chan Zuckerberg Initiative.",
"tags": [
"scrna"
],
"tasks": [],
"modalities": [
"Single Cell"
],
"organism": [],
"api": false
},
{
"id": "davis_kinase_inhibitors_db",
"name": "Davis kinase inhibitors DB",
"type": "database",
"url": "http://staff.cs.utu.fi/~aijrinas/dti/",
"description": "Experimental kinase inhibitor binding affinity dataset for proteinโ€“ligand interaction research.",
"tags": [
"chemical-protein-interaction",
"interaction"
],
"tasks": [],
"modalities": [
"Protein",
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "dependency_map_depmap",
"name": "Dependency Map (DepMap)",
"type": "database",
"url": "https://depmap.org/portal/",
"description": "CRISPR-Cas9 screens in cancer cell lines.",
"tags": [
"genome"
],
"tasks": [],
"modalities": [
"Genomics"
],
"organism": [],
"api": false
},
{
"id": "dgidb",
"name": "DGIdb",
"type": "database",
"url": "https://www.dgidb.org/",
"description": "Drug-gene interactions and the druggable genome.",
"tags": [
"drug-gene-interaction",
"interaction"
],
"tasks": [],
"modalities": [
"Gene",
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "diseases",
"name": "DISEASES",
"type": "database",
"url": "https://diseases.jensenlab.org/",
"description": "Geneโ€“disease association database integrating evidence from text mining, curated databases, and experimental data.",
"tags": [
"disease"
],
"tasks": [],
"modalities": [
"Disease"
],
"organism": [],
"api": false
},
{
"id": "disgenet",
"name": "DisGeNET",
"type": "database",
"url": "https://www.disgenet.org/",
"description": "Database of gene-disease associations integrating expert-curated and GWAS data.",
"tags": [
"disease"
],
"tasks": [],
"modalities": [
"Disease"
],
"organism": [],
"api": false
},
{
"id": "drkg",
"name": "DRKG",
"type": "database",
"url": "https://github.com/gnn4dr/DRKG",
"description": "Large-scale biological knowledge graph for drug discovery.",
"tags": [
"interaction",
"knowledge-graph"
],
"tasks": [],
"modalities": [
"Knowledge Graph"
],
"organism": [],
"api": false
},
{
"id": "drug_mechanism_database_drugmechdb",
"name": "Drug Mechanism Database (DrugMechDB)",
"type": "database",
"url": "https://github.com/SuLab/DrugMechDB/tree/2.0.1",
"description": "Mechanisms of action from drug to disease.",
"tags": [
"interaction",
"knowledge-graph"
],
"tasks": [],
"modalities": [
"Knowledge Graph"
],
"organism": [],
"api": false
},
{
"id": "drug_repurposing_hub",
"name": "Drug Repurposing Hub",
"type": "database",
"url": "https://repo-hub.broadinstitute.org/repurposing#download-data",
"description": "Collections of drug repurposing data (drug, MoA, target, etc).",
"tags": [
"compound"
],
"tasks": [],
"modalities": [
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "drugbank",
"name": "DrugBank",
"type": "database",
"url": "https://go.drugbank.com/",
"description": "Database of drugs and targets (University of Alberta).",
"tags": [
"disease"
],
"tasks": [],
"modalities": [
"chemical-structure"
],
"organism": [],
"api": false,
"entities": [
"disease",
"drug",
"protein"
],
"last_checked": "2026-08-08",
"metadata_sources": [
"https://go.drugbank.com/"
]
},
{
"id": "drugcentral",
"name": "DrugCentral",
"type": "database",
"url": "http://drugcentral.org/",
"description": "Online drug compendium with drug mode of action and indication information.",
"tags": [
"compound"
],
"tasks": [],
"modalities": [
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "drugtargetcommons",
"name": "DrugTargetCommons",
"type": "database",
"url": "https://drugtargetcommons.fimm.fi/",
"description": "Community platform for curating and integrating experimental bioactivity data across drugs and targets.",
"tags": [
"compound"
],
"tasks": [],
"modalities": [
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "encode",
"name": "ENCODE",
"type": "database",
"url": "https://www.encodeproject.org/",
"description": "Encyclopedia of DNA Elements; regulatory and functional genomic elements across the genome.",
"tags": [
"genome"
],
"tasks": [],
"modalities": [
"Genomics"
],
"organism": [],
"api": false
},
{
"id": "ensembl",
"name": "Ensembl",
"type": "database",
"url": "https://www.ensembl.org/",
"description": "Genome browser and annotation database for vertebrate and other eukaryotic genomes.",
"tags": [
"genome"
],
"tasks": [],
"modalities": [
"Genomics"
],
"organism": [],
"api": false
},
{
"id": "eu_drug_regulating_authorities_clinical_trials_db_eudract",
"name": "EU Drug Regulating Authorities Clinical Trials DB (EudraCT)",
"type": "database",
"url": "https://eudract.ema.europa.eu/",
"description": "European clinical trial database.",
"tags": [
"clinical-trial"
],
"tasks": [],
"modalities": [
"Clinical"
],
"organism": [],
"api": false
},
{
"id": "fantom5",
"name": "FANTOM5",
"type": "database",
"url": "https://fantom.gsc.riken.jp/5/",
"description": "Functional annotation of mammalian genome; comprehensive atlas of active enhancers, promoters, and transcription start sites across human and mouse cell types.",
"tags": [
"genome"
],
"tasks": [],
"modalities": [
"Genomics"
],
"organism": [],
"api": false
},
{
"id": "genbank",
"name": "GenBank",
"type": "database",
"url": "https://www.ncbi.nlm.nih.gov/genbank/",
"description": "NCBI's database of genetic sequences.",
"tags": [
"genome"
],
"tasks": [],
"modalities": [
"Genomics"
],
"organism": [],
"api": false
},
{
"id": "gene_expression_omnibus",
"name": "Gene Expression Omnibus",
"type": "database",
"url": "https://www.ncbi.nlm.nih.gov/geo/",
"description": "Public functional genomics database.",
"tags": [
"scrna"
],
"tasks": [],
"modalities": [
"Single Cell"
],
"organism": [],
"api": false
},
{
"id": "genomics_of_drug_sensitivity_in_cancer_gdsc",
"name": "Genomics of Drug Sensitivity in Cancer (GDSC)",
"type": "database",
"url": "https://www.cancerrxgene.org/",
"description": "Drug sensitivity for ~1000 human cancer cell lines and hundreds of compounds.",
"tags": [
"benchmarks-and-datasets",
"drug-cell-line-response",
"interaction"
],
"tasks": [
"drug-response-prediction"
],
"modalities": [
"genomics"
],
"organism": [],
"api": false,
"entities": [
"cell",
"drug",
"gene"
],
"last_checked": "2026-08-08",
"metadata_sources": [
"https://www.cancerrxgene.org/"
]
},
{
"id": "gnomad",
"name": "gnomAD",
"type": "database",
"url": "https://gnomad.broadinstitute.org/",
"description": "Genome Aggregation Database; genetic variation from large-scale sequencing projects.",
"tags": [
"genome"
],
"tasks": [],
"modalities": [
"Genomics"
],
"organism": [],
"api": false
},
{
"id": "hetionet",
"name": "Hetionet",
"type": "database",
"url": "https://github.com/hetio/hetionet",
"description": "Heterogeneous network integrating genes, diseases, drugs, pathways, and more.",
"tags": [
"interaction",
"knowledge-graph"
],
"tasks": [],
"modalities": [
"Knowledge Graph"
],
"organism": [],
"api": false
},
{
"id": "hippie",
"name": "HIPPIE",
"type": "database",
"url": "http://cbdm-01.zdv.uni-mainz.de/~mschaefer/hippie/",
"description": "Human protein-protein interaction database.",
"tags": [
"interaction",
"protein-protein-interaction"
],
"tasks": [],
"modalities": [
"Protein"
],
"organism": [],
"api": false
},
{
"id": "hmdb_human_metabolome_database",
"name": "HMDB (Human Metabolome Database)",
"type": "database",
"url": "https://hmdb.ca/",
"description": "Comprehensive database of small molecule metabolites found in the human body.",
"tags": [
"compound"
],
"tasks": [],
"modalities": [
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "human_cell_atlas",
"name": "Human Cell Atlas",
"type": "database",
"url": "https://www.humancellatlas.org/",
"description": "Open global atlas of all cells in the human body.",
"tags": [
"scrna"
],
"tasks": [],
"modalities": [
"Single Cell"
],
"organism": [],
"api": false
},
{
"id": "human_genome_resources_at_ncbi",
"name": "Human Genome Resources at NCBI",
"type": "database",
"url": "https://www.ncbi.nlm.nih.gov/projects/genome/guide/human/index.shtml",
"description": "Database for genomics, proteomics, transcriptomics, and systems biology.",
"tags": [
"genome"
],
"tasks": [],
"modalities": [
"Genomics"
],
"organism": [],
"api": false
},
{
"id": "human_phenotype_ontology_hpo",
"name": "Human Phenotype Ontology (HPO)",
"type": "database",
"url": "https://hpo.jax.org/",
"description": "Standardized vocabulary of phenotypic abnormalities in human disease, linking genes, variants, and clinical features.",
"tags": [
"disease"
],
"tasks": [],
"modalities": [
"Disease"
],
"organism": [],
"api": false
},
{
"id": "icd10",
"name": "ICD10",
"type": "database",
"url": "https://icd.who.int/browse10/2019/en",
"description": "International Classification of Diseases, 10th revision.",
"tags": [
"clinical-trial"
],
"tasks": [],
"modalities": [
"Clinical"
],
"organism": [],
"api": false
},
{
"id": "intact",
"name": "IntAct",
"type": "database",
"url": "https://www.ebi.ac.uk/intact/home",
"description": "Open-source molecular interaction database and analysis system from EMBL-EBI.",
"tags": [
"interaction",
"protein-protein-interaction"
],
"tasks": [],
"modalities": [
"Protein"
],
"organism": [],
"api": false
},
{
"id": "interpro",
"name": "InterPro",
"type": "database",
"url": "https://www.ebi.ac.uk/interpro/",
"description": "Protein families, domains, and functional sites database integrating 14 member databases including Pfam and PROSITE.",
"tags": [
"protein"
],
"tasks": [],
"modalities": [
"Protein"
],
"organism": [],
"api": false
},
{
"id": "jaspar",
"name": "JASPAR",
"type": "database",
"url": "http://jaspar.genereg.net/",
"description": "Database of transcription factor binding profiles.",
"tags": [
"genome"
],
"tasks": [],
"modalities": [
"Genomics"
],
"organism": [],
"api": false
},
{
"id": "kegg_compound",
"name": "KEGG COMPOUND",
"type": "database",
"url": "https://www.genome.jp/kegg/compound/",
"description": "Collection of small molecules and biopolymers.",
"tags": [
"compound"
],
"tasks": [],
"modalities": [
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "kegg_drug",
"name": "KEGG DRUG",
"type": "database",
"url": "https://www.genome.jp/kegg/drug/",
"description": "Comprehensive, approved drug information.",
"tags": [
"disease"
],
"tasks": [],
"modalities": [
"Disease"
],
"organism": [],
"api": false
},
{
"id": "kegg_pathway",
"name": "KEGG PATHWAY",
"type": "database",
"url": "https://www.genome.jp/kegg/pathway.html",
"description": "Collection of pathway maps.",
"tags": [
"pathway"
],
"tasks": [],
"modalities": [
"Pathway"
],
"organism": [],
"api": false
},
{
"id": "kinase_inhibitor_bioactivity_data_kiba",
"name": "Kinase Inhibitor Bioactivity Data (KIBA)",
"type": "database",
"url": "https://janeliascicomp.github.io/KIBA/",
"description": "Integrated bioactivity scores for kinase inhibitors combining Ki, Kd, and IC50 measurements.",
"tags": [
"chemical-protein-interaction",
"interaction"
],
"tasks": [],
"modalities": [
"Protein",
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "lipid_maps",
"name": "LIPID MAPS",
"type": "database",
"url": "https://www.lipidmaps.org/databases/lmsd/overview",
"description": "Database of lipids.",
"tags": [
"compound"
],
"tasks": [],
"modalities": [
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "massbank",
"name": "MassBank",
"type": "database",
"url": "http://www.massbank.jp/",
"description": "Open source databases and tools for mass spectrometry reference spectra.",
"tags": [
"mass-spectra"
],
"tasks": [],
"modalities": [
"Mass Spectra"
],
"organism": [],
"api": false
},
{
"id": "mgnify",
"name": "MGnify",
"type": "database",
"url": "https://www.ebi.ac.uk/metagenomics/",
"description": "Resource for metagenomic and metatranscriptomic data.",
"tags": [
"genome"
],
"tasks": [],
"modalities": [
"Genomics"
],
"organism": [],
"api": false
},
{
"id": "mimic_iv",
"name": "MIMIC-IV",
"type": "database",
"url": "https://mimic.mit.edu/",
"description": "Freely accessible critical care database.",
"tags": [
"clinical-trial"
],
"tasks": [],
"modalities": [
"Clinical"
],
"organism": [],
"api": false
},
{
"id": "mirbase",
"name": "miRBase",
"type": "database",
"url": "https://www.mirbase.org/",
"description": "Reference repository for microRNA gene annotations, sequences, and experimentally validated targets.",
"tags": [
"gene-regulatory-network",
"interaction"
],
"tasks": [],
"modalities": [
"Gene Expression"
],
"organism": [],
"api": false
},
{
"id": "mona_massbank_of_north_america",
"name": "MoNA MassBank of North America",
"type": "database",
"url": "https://mona.fiehnlab.ucdavis.edu/",
"description": "Meta-database of metabolite mass spectra, metadata, and associated compounds.",
"tags": [
"mass-spectra"
],
"tasks": [],
"modalities": [
"Mass Spectra"
],
"organism": [],
"api": false
},
{
"id": "msigdb_molecular_signatures_database",
"name": "MSigDB (Molecular Signatures Database)",
"type": "database",
"url": "https://www.gsea-msigdb.org/gsea/msigdb",
"description": "Curated gene sets derived from pathways and biological processes.",
"tags": [
"pathway"
],
"tasks": [],
"modalities": [
"Pathway"
],
"organism": [],
"api": false
},
{
"id": "nci60",
"name": "NCI60",
"type": "database",
"url": "https://dtp.cancer.gov/discovery_development/nci-60/",
"description": "Focuses on 60 cancer cell lines and many drugs.",
"tags": [
"benchmarks-and-datasets",
"drug-cell-line-response",
"interaction"
],
"tasks": [],
"modalities": [
"Gene Expression",
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "nextprot",
"name": "NeXtProt",
"type": "database",
"url": "https://www.nextprot.org/",
"description": "Expert knowledge base on human proteins with deep functional annotation, complementary to UniProt.",
"tags": [
"protein"
],
"tasks": [],
"modalities": [
"Protein"
],
"organism": [],
"api": false
},
{
"id": "oadb_observed_antibody_space_database",
"name": "OADB (Observed Antibody Space Database)",
"type": "database",
"url": "http://opig.stats.ox.ac.uk/webapps/oas/",
"description": "Database of antibody sequences from immune repertoire sequencing.",
"tags": [
"protein"
],
"tasks": [],
"modalities": [
"Protein"
],
"organism": [],
"api": false
},
{
"id": "omim_online_mendelian_inheritance_in_man",
"name": "OMIM (Online Mendelian Inheritance in Man)",
"type": "database",
"url": "https://www.omim.org/",
"description": "Comprehensive database of human genes and genetic disorders.",
"tags": [
"disease"
],
"tasks": [],
"modalities": [
"Disease"
],
"organism": [],
"api": false
},
{
"id": "omnipath",
"name": "OmniPath",
"type": "database",
"url": "https://omnipathdb.org/",
"description": "Comprehensive resource integrating protein interactions, signaling pathways, gene regulatory networks, and miRNA targets from over 100 databases.",
"tags": [
"pathway"
],
"tasks": [],
"modalities": [
"Pathway"
],
"organism": [],
"api": false
},
{
"id": "open_targets_platform",
"name": "Open Targets Platform",
"type": "database",
"url": "https://platform.opentargets.org/",
"description": "Systematic target identification and prioritization platform integrating genetics, genomics, and drug data for drug discovery.",
"tags": [
"disease"
],
"tasks": [],
"modalities": [
"Disease"
],
"organism": [],
"api": false
},
{
"id": "pathwaycommons",
"name": "PathwayCommons",
"type": "database",
"url": "https://www.pathwaycommons.org/",
"description": "Database of pathways and interactions.",
"tags": [
"pathway"
],
"tasks": [],
"modalities": [
"Pathway"
],
"organism": [],
"api": false
},
{
"id": "pdbbind",
"name": "PDBBind",
"type": "database",
"url": "https://www.pdbbind-plus.org.cn/",
"description": "Binding affinity data for biomolecular complexes.",
"tags": [
"chemical-protein-interaction",
"interaction"
],
"tasks": [],
"modalities": [
"Protein",
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "pfam",
"name": "Pfam",
"type": "database",
"url": "https://www.ebi.ac.uk/interpro/entry/pfam/",
"description": "Database of protein families described by multiple sequence alignments and hidden Markov models.",
"tags": [
"protein"
],
"tasks": [],
"modalities": [
"Protein"
],
"organism": [],
"api": false
},
{
"id": "primekg",
"name": "PrimeKG",
"type": "database",
"url": "https://github.com/mims-harvard/PrimeKG",
"description": "Multi-modal precision medicine knowledge graph integrating clinical, genetic, and drug data.",
"tags": [
"interaction",
"knowledge-graph"
],
"tasks": [],
"modalities": [
"Knowledge Graph"
],
"organism": [],
"api": false
},
{
"id": "protein_data_bank_pdb",
"name": "PROTEIN DATA BANK (PDB)",
"type": "database",
"url": "https://www.rcsb.org/",
"description": "3D structures of proteins, nucleic acids, complexes.",
"tags": [
"protein"
],
"tasks": [],
"modalities": [
"Protein"
],
"organism": [],
"api": false
},
{
"id": "pubchem",
"name": "PubChem",
"type": "database",
"url": "https://pubchem.ncbi.nlm.nih.gov/",
"description": "One of the largest chemical databases (compounds, genes, and proteins).",
"tags": [
"compound"
],
"tasks": [],
"modalities": [
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "rcsb_protein_data_bank",
"name": "RCSB Protein Data Bank",
"type": "database",
"url": "https://www.rcsb.org/",
"description": "Repository for structural data of biological molecules.",
"tags": [
"protein"
],
"tasks": [],
"modalities": [
"Protein"
],
"organism": [],
"api": false
},
{
"id": "reactome",
"name": "Reactome",
"type": "database",
"url": "https://reactome.org/",
"description": "Expert-curated, peer-reviewed pathway database with detailed reaction mechanisms.",
"tags": [
"pathway"
],
"tasks": [],
"modalities": [
"Pathway"
],
"organism": [],
"api": false
},
{
"id": "regnetwork",
"name": "RegNetwork",
"type": "database",
"url": "http://www.regnetworkweb.org/",
"description": "Database of gene regulatory networks covering transcription factorโ€“target gene and miRNAโ€“gene interaction data across multiple species.",
"tags": [
"gene-regulatory-network",
"interaction"
],
"tasks": [],
"modalities": [
"Gene Expression"
],
"organism": [],
"api": false
},
{
"id": "rfam",
"name": "Rfam",
"type": "database",
"url": "https://rfam.org/",
"description": "Database of RNA families with sequence alignments and consensus structures.",
"tags": [
"genome"
],
"tasks": [],
"modalities": [
"Genomics"
],
"organism": [],
"api": false
},
{
"id": "rhea",
"name": "Rhea",
"type": "database",
"url": "https://www.rhea-db.org/",
"description": "Database of chemical reactions.",
"tags": [
"compound"
],
"tasks": [],
"modalities": [
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "roadmap_epigenomics",
"name": "ROADMAP Epigenomics",
"type": "database",
"url": "http://www.roadmapepigenomics.org/",
"description": "Reference epigenome maps for 111 primary human cell types and tissues, including histone modifications, chromatin accessibility, and DNA methylation.",
"tags": [
"genome"
],
"tasks": [],
"modalities": [
"Genomics"
],
"organism": [],
"api": false
},
{
"id": "sabdab",
"name": "SAbDab",
"type": "database",
"url": "https://opig.stats.ox.ac.uk/webapps/sabdab-sabpred/sabdab",
"description": "Structural Antibody Database containing all antibody structures in the PDB.",
"tags": [
"protein"
],
"tasks": [],
"modalities": [
"Protein"
],
"organism": [],
"api": false
},
{
"id": "signor_2_0",
"name": "SIGNOR 2.0",
"type": "database",
"url": "https://signor.uniroma2.it/",
"description": "Database of causal signaling interactions and pathways, with signed and directed relationships between proteins.",
"tags": [
"pathway"
],
"tasks": [],
"modalities": [
"Pathway"
],
"organism": [],
"api": false
},
{
"id": "single_cell_expression_atlas",
"name": "Single Cell Expression Atlas",
"type": "database",
"url": "https://www.ebi.ac.uk/gxa/sc/home",
"description": "Public database for single-cell RNA.",
"tags": [
"scrna"
],
"tasks": [],
"modalities": [
"Single Cell"
],
"organism": [],
"api": false
},
{
"id": "single_cell_portal",
"name": "Single Cell PORTAL",
"type": "database",
"url": "https://singlecell.broadinstitute.org/single_cell",
"description": "Public database for single-cell RNA.",
"tags": [
"scrna"
],
"tasks": [],
"modalities": [
"Single Cell"
],
"organism": [],
"api": false
},
{
"id": "snap",
"name": "SNAP",
"type": "database",
"url": "https://snap.stanford.edu/biodata/datasets/10002/10002-ChG-Miner.html",
"description": "Dataset of drug-gene interactions.",
"tags": [
"drug-gene-interaction",
"interaction"
],
"tasks": [],
"modalities": [
"Gene",
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "stitch",
"name": "STITCH",
"type": "database",
"url": "http://stitch.embl.de/",
"description": "Chemical-protein interactions.",
"tags": [
"chemical-protein-interaction",
"interaction"
],
"tasks": [],
"modalities": [
"Protein",
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "string",
"name": "STRING",
"type": "database",
"url": "https://string-db.org/",
"description": "PPI networks for multiple organisms.",
"tags": [
"interaction",
"protein-protein-interaction"
],
"tasks": [],
"modalities": [
"knowledge-graph",
"proteomics"
],
"organism": [],
"api": false,
"entities": [
"protein"
],
"documentation": "https://string-db.org/help/api/",
"last_checked": "2026-08-08",
"metadata_sources": [
"https://string-db.org/",
"https://string-db.org/help/api/"
]
},
{
"id": "the_genotype_tissue_expression_gtex",
"name": "The Genotype-Tissue Expression (GTEx)",
"type": "database",
"url": "https://gtexportal.org/home/",
"description": "Human gene expression and regulation resource.",
"tags": [
"genome"
],
"tasks": [],
"modalities": [
"Genomics"
],
"organism": [],
"api": false
},
{
"id": "the_human_protein_atlas",
"name": "THE HUMAN PROTEIN ATLAS",
"type": "database",
"url": "https://www.proteinatlas.org/",
"description": "Comprehensive human protein database (cells, tissues, organs).",
"tags": [
"protein"
],
"tasks": [],
"modalities": [
"Protein"
],
"organism": [],
"api": false
},
{
"id": "therapeutic_target_database",
"name": "Therapeutic Target Database",
"type": "database",
"url": "https://idrblab.net/ttd/full-data-download",
"description": "Drug-target, target-disease, and drug-disease datasets.",
"tags": [
"compound"
],
"tasks": [],
"modalities": [
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "trrust_v2",
"name": "TRRUST v2",
"type": "database",
"url": "https://www.grnpedia.org/trrust/",
"description": "Manually curated database of human and mouse transcriptional regulatory interactions between transcription factors and their target genes, expanded with literature-derived evidence.",
"tags": [
"gene-regulatory-network",
"interaction"
],
"tasks": [],
"modalities": [
"Gene Expression"
],
"organism": [],
"api": false
},
{
"id": "ucsc_genome_browser",
"name": "UCSC Genome Browser",
"type": "database",
"url": "https://genome.ucsc.edu/",
"description": "UCSC's genome browser.",
"tags": [
"genome"
],
"tasks": [],
"modalities": [
"Genomics"
],
"organism": [],
"api": false
},
{
"id": "uniclust",
"name": "Uniclust",
"type": "database",
"url": "https://uniclust.mmseqs.com/",
"description": "Clustered protein sequence databases.",
"tags": [
"protein"
],
"tasks": [],
"modalities": [
"Protein"
],
"organism": [],
"api": false
},
{
"id": "uniprot",
"name": "UniProt",
"type": "database",
"url": "https://www.uniprot.org/",
"description": "Functional information on proteins.",
"tags": [
"protein"
],
"tasks": [],
"modalities": [
"Protein"
],
"organism": [],
"api": false
},
{
"id": "uniref",
"name": "UniRef",
"type": "database",
"url": "https://www.uniprot.org/uniref/",
"description": "Non-redundant sequence database clustering UniProtKB entries at multiple sequence identity thresholds.",
"tags": [
"protein"
],
"tasks": [],
"modalities": [
"Protein"
],
"organism": [],
"api": false
},
{
"id": "wikipathways",
"name": "WikiPathways",
"type": "database",
"url": "https://wikipathways.org/",
"description": "Database of biological pathways.",
"tags": [
"pathway"
],
"tasks": [],
"modalities": [
"Pathway"
],
"organism": [],
"api": false
},
{
"id": "zinc_ligand_discovery_database",
"name": "ZINC ligand discovery database",
"type": "database",
"url": "https://zinc.docking.org/",
"description": "Free database of commercially-available compounds for virtual screening.",
"tags": [
"compound"
],
"tasks": [],
"modalities": [
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "aestetik",
"name": "AESTETIK",
"type": "model",
"url": "https://github.com/ratschlab/aestetik",
"description": "Autoencoder for spatial transcriptomics representation learning using topology and histology image knowledge.",
"tags": [
"foundation-models",
"single-cell-foundation-models",
"spatial-foundation-models"
],
"tasks": [
"representation-learning"
],
"modalities": [
"histopathology",
"spatial-transcriptomics"
],
"organism": [],
"api": false,
"entities": [
"cell",
"gene",
"tissue"
],
"methods": [
"autoencoder"
],
"github": "https://github.com/ratschlab/aestetik",
"last_checked": "2026-08-08",
"metadata_sources": [
"https://github.com/ratschlab/aestetik"
]
},
{
"id": "ai4chem_chemllm_7b_chat",
"name": "AI4Chem/ChemLLM-7B-Chat",
"type": "model",
"url": "https://huggingface.co/AI4Chem/ChemLLM-7B-Chat",
"description": "LLM for chemical & molecular science.",
"tags": [
"llm-for-biology"
],
"tasks": [
"Language Modeling"
],
"modalities": [
"Text"
],
"organism": [],
"api": false
},
{
"id": "alphafold3",
"name": "AlphaFold3",
"type": "model",
"url": "https://github.com/google-deepmind/alphafold3",
"description": "Predicts structures of proteins, nucleic acids, small molecules, and their complexes.",
"tags": [
"foundation-models",
"protein-foundation-models",
"protein-structure-prediction-and-design"
],
"tasks": [
"structure-prediction"
],
"modalities": [
"molecular-structure",
"protein-sequence"
],
"organism": [],
"api": false,
"entities": [
"molecule",
"protein",
"protein-complex"
],
"methods": [
"diffusion"
],
"github": "https://github.com/google-deepmind/alphafold3",
"last_checked": "2026-08-08",
"metadata_sources": [
"https://github.com/google-deepmind/alphafold3"
]
},
{
"id": "ankh",
"name": "Ankh",
"type": "model",
"url": "https://github.com/agemagician/Ankh",
"description": "Efficient protein language model optimized for downstream prediction tasks including secondary structure, localization, and function annotation.",
"tags": [
"foundation-models",
"pre-trained-embedding",
"protein-foundation-models"
],
"tasks": [
"Foundation Model"
],
"modalities": [
"Protein"
],
"organism": [],
"api": false
},
{
"id": "babel",
"name": "BABEL",
"type": "model",
"url": "https://github.com/wukevin/babel",
"description": "Cross-modality translation model enabling prediction between scRNA-seq and scATAC-seq profiles without requiring paired single-cell measurements.",
"tags": [
"foundation-models",
"multi-omics-foundation-models",
"single-cell-foundation-models"
],
"tasks": [
"Foundation Model"
],
"modalities": [
"Multi-Omics",
"Single Cell"
],
"organism": [],
"api": false
},
{
"id": "basenji",
"name": "Basenji",
"type": "model",
"url": "https://github.com/calico/basenji",
"description": "Sequential regulatory activity prediction from DNA sequences.",
"tags": [
"foundation-models",
"genomics-foundation-models"
],
"tasks": [
"Foundation Model"
],
"modalities": [
"Genomics"
],
"organism": [],
"api": false
},
{
"id": "biogpt",
"name": "BioGPT",
"type": "model",
"url": "https://github.com/microsoft/BioGPT",
"description": "LLM for biomedical text generation.",
"tags": [
"llm-for-biology"
],
"tasks": [
"Language Modeling"
],
"modalities": [
"Text"
],
"organism": [],
"api": false
},
{
"id": "biomedclip",
"name": "BiomedCLIP",
"type": "model",
"url": "https://huggingface.co/microsoft/BiomedCLIP-PubMedBERT_256-vit_g_14",
"description": "CLIP-based vision-language foundation model for biomedical images and text trained on PubMed figureโ€“caption pairs.",
"tags": [
"foundation-models",
"multi-modal-foundation-models"
],
"tasks": [
"Foundation Model"
],
"modalities": [
"Multi-Modal"
],
"organism": [],
"api": false
},
{
"id": "biomedlm",
"name": "BioMedLM",
"type": "model",
"url": "https://huggingface.co/stanford-crfm/BioMedLM",
"description": "2.7B parameter GPT-2-style language model trained exclusively on biomedical literature from PubMed for biomedical question answering and text generation.",
"tags": [
"llm-for-biology"
],
"tasks": [
"Language Modeling"
],
"modalities": [
"Text"
],
"organism": [],
"api": false
},
{
"id": "boltz_1",
"name": "Boltz-1",
"type": "model",
"url": "https://github.com/jwohlwend/boltz",
"description": "Open-source all-atom biomolecular structure prediction model for proteins, nucleic acids, small molecules, and their complexes achieving AlphaFold3-level accuracy.",
"tags": [
"foundation-models",
"protein-foundation-models",
"protein-structure-prediction-and-design"
],
"tasks": [
"Foundation Model",
"Protein Structure Prediction"
],
"modalities": [
"Protein"
],
"organism": [],
"api": false
},
{
"id": "borzoi",
"name": "Borzoi",
"type": "model",
"url": "https://github.com/calico/borzoi",
"description": "Extended successor to Enformer for predicting RNA-seq coverage from long genomic sequence windows (524 kb) with improved resolution.",
"tags": [
"foundation-models",
"genomics-foundation-models"
],
"tasks": [
"Foundation Model"
],
"modalities": [
"Genomics"
],
"organism": [],
"api": false
},
{
"id": "bulkformer",
"name": "BulkFormer",
"type": "model",
"url": "https://github.com/KangBoming/BulkFormer",
"description": "Foundation model for bulk RNA-seq data; learns general transcriptomic representations.",
"tags": [
"foundation-models",
"single-cell-foundation-models",
"transcriptomics-foundation-models"
],
"tasks": [
"Foundation Model"
],
"modalities": [
"Single Cell",
"Transcriptomics"
],
"organism": [],
"api": false
},
{
"id": "caduceus",
"name": "Caduceus",
"type": "model",
"url": "https://github.com/kuleshov-group/caduceus",
"description": "Bidirectional equivariant long-range DNA sequence model based on Mamba.",
"tags": [
"foundation-models",
"genomics-foundation-models"
],
"tasks": [
"Foundation Model"
],
"modalities": [
"Genomics"
],
"organism": [],
"api": false
},
{
"id": "cancerfoundation",
"name": "CancerFoundation",
"type": "model",
"url": "https://github.com/BoevaLab/CancerFoundation",
"description": "Single-cell RNA-seq foundation model trained exclusively on a curated dataset of malignant cells to learn cancer-specific embeddings.",
"tags": [
"foundation-models",
"single-cell-foundation-models",
"transcriptomics-foundation-models"
],
"tasks": [
"Foundation Model"
],
"modalities": [
"Single Cell",
"Transcriptomics"
],
"organism": [],
"api": false
},
{
"id": "cassia",
"name": "CASSIA",
"type": "model",
"url": "https://github.com/ElliotXie/CASSIA",
"description": "Multi-agent LLM for reference-free, interpretable cell-type annotation of single-cell RNA-seq data, with dedicated annotation, validation, scoring, and reporting agents.",
"tags": [
"llm-for-biology"
],
"tasks": [
"Language Modeling"
],
"modalities": [
"Text"
],
"organism": [],
"api": false
},
{
"id": "cellot",
"name": "CellOT",
"type": "model",
"url": "https://github.com/bunnech/cellot",
"description": "Neural optimal transport framework for predicting single-cell responses to drug and genetic perturbations.",
"tags": [
"drug-discovery",
"drug-perturbation"
],
"tasks": [
"Drug Discovery",
"Drug Perturbation"
],
"modalities": [
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "cellplm",
"name": "CellPLM",
"type": "model",
"url": "https://github.com/OmicsML/CellPLM",
"description": "Cell pre-trained language model with inter-cell transformer architecture for diverse single-cell analysis tasks.",
"tags": [
"foundation-models",
"single-cell-foundation-models",
"transcriptomics-foundation-models"
],
"tasks": [
"foundation-model-pretraining",
"representation-learning"
],
"modalities": [
"single-cell-rna-seq",
"transcriptomics"
],
"organism": [],
"api": false,
"entities": [
"cell",
"gene"
],
"methods": [
"self-supervised-learning",
"transformer"
],
"year": 2023,
"github": "https://github.com/OmicsML/CellPLM",
"paper": "https://www.biorxiv.org/content/10.1101/2023.10.03.560734v1",
"last_checked": "2026-08-08",
"metadata_sources": [
"https://github.com/OmicsML/CellPLM",
"https://www.biorxiv.org/content/10.1101/2023.10.03.560734v1"
]
},
{
"id": "chai_1",
"name": "Chai-1",
"type": "model",
"url": "https://github.com/chaidiscovery/chai-lab",
"description": "Unified molecular structure prediction model covering proteins, nucleic acids, small molecules, and complexes.",
"tags": [
"foundation-models",
"protein-foundation-models",
"protein-structure-prediction-and-design"
],
"tasks": [
"Foundation Model",
"Protein Structure Prediction"
],
"modalities": [
"Protein"
],
"organism": [],
"api": false
},
{
"id": "chatdrug",
"name": "ChatDrug",
"type": "model",
"url": "https://github.com/chao1224/ChatDrug",
"description": "LLM-based conversational pipeline for drug discovery, using natural language prompts for iterative drug editing and optimization.",
"tags": [
"llm-for-biology"
],
"tasks": [
"Language Modeling"
],
"modalities": [
"Text"
],
"organism": [],
"api": false
},
{
"id": "chemberta_2",
"name": "ChemBERTa-2",
"type": "model",
"url": "https://github.com/seyonechithrananda/bert-loves-chemistry",
"description": "RoBERTa-based molecular language model pretrained on SMILES for small-molecule representation learning.",
"tags": [
"compound-embedding",
"compound-foundation-models",
"foundation-models"
],
"tasks": [
"representation-learning"
],
"modalities": [
"chemical-structure"
],
"organism": [],
"api": false,
"entities": [
"molecule"
],
"methods": [
"language-model",
"self-supervised-learning",
"transformer"
],
"github": "https://github.com/seyonechithrananda/bert-loves-chemistry",
"last_checked": "2026-08-08",
"metadata_sources": [
"https://github.com/seyonechithrananda/bert-loves-chemistry"
]
},
{
"id": "chemcpa",
"name": "chemCPA",
"type": "model",
"url": "https://github.com/theislab/chemCPA",
"description": "Compositional perturbation autoencoder for predicting single-cell transcriptional responses to unseen drug perturbations and dose combinations.",
"tags": [
"drug-discovery",
"drug-perturbation"
],
"tasks": [
"Drug Discovery",
"Drug Perturbation"
],
"modalities": [
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "chief",
"name": "CHIEF",
"type": "model",
"url": "https://github.com/hms-dbmi/CHIEF",
"description": "Clinical Histopathology Imaging Evaluation Foundation model integrating histology images and clinical context for pan-cancer analysis.",
"tags": [
"foundation-models",
"multi-modal-foundation-models"
],
"tasks": [
"Foundation Model"
],
"modalities": [
"Multi-Modal"
],
"organism": [],
"api": false
},
{
"id": "clawbio",
"name": "ClawBio",
"type": "model",
"url": "https://github.com/ClawBio/ClawBio",
"description": "Bioinformatics-native AI agent skill library with local-first pharmacogenomics, ancestry PCA, semantic similarity, nutrigenomics, and metagenomics skills.",
"tags": [
"llm-for-biology"
],
"tasks": [
"Language Modeling"
],
"modalities": [
"Text"
],
"organism": [],
"api": false
},
{
"id": "cmonge",
"name": "CMonge",
"type": "model",
"url": "https://github.com/AI4SCR/conditional-monge-gap",
"description": "Conditional optimal transport model for generalizable single-cell perturbation response prediction across drugs and doses.",
"tags": [
"drug-discovery",
"drug-perturbation"
],
"tasks": [
"Drug Discovery",
"Drug Perturbation"
],
"modalities": [
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "concerto",
"name": "Concerto",
"type": "model",
"url": "https://github.com/melobio/Concerto-reproducibility",
"description": "Contrastive self-supervised learning framework for single-cell multimodal data integration, batch correction, and reference-query mapping.",
"tags": [
"foundation-models",
"multi-omics-foundation-models",
"single-cell-foundation-models"
],
"tasks": [
"Foundation Model"
],
"modalities": [
"Multi-Omics",
"Single Cell"
],
"organism": [],
"api": false
},
{
"id": "conch",
"name": "CONCH",
"type": "model",
"url": "https://github.com/mahmoodlab/CONCH",
"description": "Vision-language foundation model for computational pathology trained with contrastive captioning on pathology imageโ€“text pairs.",
"tags": [
"foundation-models",
"single-cell-foundation-models",
"spatial-foundation-models"
],
"tasks": [
"foundation-model-pretraining",
"representation-learning"
],
"modalities": [
"histopathology",
"imaging"
],
"organism": [],
"api": false,
"entities": [
"tissue"
],
"methods": [
"contrastive-learning",
"transformer"
],
"github": "https://github.com/mahmoodlab/CONCH",
"last_checked": "2026-08-08",
"metadata_sources": [
"https://github.com/mahmoodlab/CONCH"
]
},
{
"id": "cyclecdr",
"name": "cycleCDR",
"type": "model",
"url": "https://github.com/hliulab/cycleCDR",
"description": "Interpretable cycle-consistency framework for modeling cellular responses to drug perturbations.",
"tags": [
"drug-discovery",
"drug-perturbation"
],
"tasks": [
"Drug Discovery",
"Drug Perturbation"
],
"modalities": [
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "deepaeg",
"name": "DeepAEG",
"type": "model",
"url": "https://github.com/zhejiangzhuque/DeepAEG",
"description": "GNN embedding + attention mechanism.",
"tags": [
"drug-discovery",
"drug-response-prediction"
],
"tasks": [
"Drug Discovery",
"Drug Response Prediction"
],
"modalities": [
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "deepdsc",
"name": "DeepDSC",
"type": "model",
"url": "https://ieeexplore-ieee-org.ezp2.lib.umn.edu/stamp/stamp.jsp?tp=&arnumber=8723620&tag=1",
"description": "Autoencoder + fully connected NN.",
"tags": [
"drug-discovery",
"drug-response-prediction"
],
"tasks": [
"Drug Discovery",
"Drug Response Prediction"
],
"modalities": [
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "deepdta",
"name": "DeepDTA",
"type": "model",
"url": "https://github.com/hkmztrk/DeepDTA",
"description": "Deep learning model using CNNs on protein sequences and drug SMILES.",
"tags": [
"drug-discovery",
"drug-target-interaction"
],
"tasks": [
"Drug Discovery",
"Drug Target Interaction"
],
"modalities": [
"Protein",
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "deeppurpose",
"name": "DeepPurpose",
"type": "model",
"url": "https://github.com/kexinhuang12345/DeepPurpose",
"description": "Deep learning library for drug repurposing.",
"tags": [
"drug-discovery",
"drug-repurposing"
],
"tasks": [
"Drug Discovery",
"Drug Repurposing"
],
"modalities": [
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "deepsea",
"name": "DeepSEA",
"type": "model",
"url": "http://deepsea.princeton.edu/",
"description": "Deep learning framework for predicting chromatin effects of sequence alterations with single-nucleotide sensitivity across thousands of chromatin features.",
"tags": [
"foundation-models",
"genomics-foundation-models"
],
"tasks": [
"Foundation Model"
],
"modalities": [
"Genomics"
],
"organism": [],
"api": false
},
{
"id": "deepspot",
"name": "DeepSpot",
"type": "model",
"url": "https://github.com/ratschlab/DeepSpot",
"description": "Deep learning model predicting spatial transcriptomics from H&E images at spot and single-cell resolution.",
"tags": [
"foundation-models",
"single-cell-foundation-models",
"spatial-foundation-models"
],
"tasks": [
"regression"
],
"modalities": [
"histopathology",
"spatial-transcriptomics"
],
"organism": [],
"api": false,
"entities": [
"gene",
"tissue"
],
"github": "https://github.com/ratschlab/DeepSpot",
"last_checked": "2026-08-08",
"metadata_sources": [
"https://github.com/ratschlab/DeepSpot"
]
},
{
"id": "deepspot_m",
"name": "DeepSpot-M",
"type": "model",
"url": "https://github.com/ratschlab/DeepSpotM",
"description": "Multimodal foundation model for transcriptome-wide virtual spatial transcriptomics from histology.",
"tags": [
"foundation-models",
"single-cell-foundation-models",
"spatial-foundation-models"
],
"tasks": [
"foundation-model-pretraining",
"regression"
],
"modalities": [
"histopathology",
"spatial-transcriptomics",
"transcriptomics"
],
"organism": [],
"api": false,
"entities": [
"gene",
"tissue"
],
"github": "https://github.com/ratschlab/DeepSpotM",
"last_checked": "2026-08-08",
"metadata_sources": [
"https://github.com/ratschlab/DeepSpotM"
]
},
{
"id": "deepspot2cell",
"name": "DeepSpot2Cell",
"type": "model",
"url": "https://github.com/ratschlab/DeepSpot2Cell",
"description": "Predicts virtual single-cell spatial transcriptomics from H&E using spot-level supervision (NeurIPS 2025 Imageomics).",
"tags": [
"foundation-models",
"single-cell-foundation-models",
"spatial-foundation-models"
],
"tasks": [
"regression"
],
"modalities": [
"histopathology",
"spatial-transcriptomics"
],
"organism": [],
"api": false,
"entities": [
"cell",
"gene",
"tissue"
],
"github": "https://github.com/ratschlab/DeepSpot2Cell",
"last_checked": "2026-08-08",
"metadata_sources": [
"https://github.com/ratschlab/DeepSpot2Cell"
]
},
{
"id": "dgdrp",
"name": "DGDRP",
"type": "model",
"url": "https://github.com/minwoopak/heteronet",
"description": "Multi-view embedding neural network.",
"tags": [
"drug-discovery",
"drug-response-prediction"
],
"tasks": [
"Drug Discovery",
"Drug Response Prediction"
],
"modalities": [
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "diffdock",
"name": "DiffDock",
"type": "model",
"url": "https://github.com/gcorso/DiffDock",
"description": "Diffusion generative model for molecular docking, predicting the binding pose of small molecules to protein targets.",
"tags": [
"drug-discovery",
"molecular-generation"
],
"tasks": [
"docking"
],
"modalities": [
"molecular-structure"
],
"organism": [],
"api": false,
"entities": [
"molecule",
"protein"
],
"methods": [
"diffusion",
"geometric-deep-learning"
],
"year": 2023,
"github": "https://github.com/gcorso/DiffDock",
"paper": "https://openreview.net/forum?id=kKF8_K-mBbS",
"last_checked": "2026-08-08",
"metadata_sources": [
"https://github.com/gcorso/DiffDock",
"https://openreview.net/forum?id=kKF8_K-mBbS"
]
},
{
"id": "diffsbdd",
"name": "DiffSBDD",
"type": "model",
"url": "https://github.com/arneschneuing/DiffSBDD",
"description": "Equivariant diffusion model for structure-based drug design that generates molecules and binding conformations for protein targets.",
"tags": [
"drug-discovery",
"molecular-generation"
],
"tasks": [
"Drug Discovery",
"Molecular Generation"
],
"modalities": [
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "dnabert",
"name": "DNABERT",
"type": "model",
"url": "https://github.com/jerryji1993/DNABERT",
"description": "Pre-trained bidirectional encoder for DNA sequence analysis.",
"tags": [
"foundation-models",
"genomics-foundation-models"
],
"tasks": [
"Foundation Model"
],
"modalities": [
"Genomics"
],
"organism": [],
"api": false
},
{
"id": "dnabert_2",
"name": "DNABERT-2",
"type": "model",
"url": "https://github.com/Zhihan1996/DNABERT_2",
"description": "Improved genome foundation model with efficient tokenization.",
"tags": [
"foundation-models",
"genomics-foundation-models"
],
"tasks": [
"Foundation Model"
],
"modalities": [
"Genomics"
],
"organism": [],
"api": false
},
{
"id": "drgat",
"name": "drGAT",
"type": "model",
"url": "https://github.com/inoue0426/drGAT",
"description": "Attention-based model for drug response prediction with gene explainability.",
"tags": [
"drug-discovery",
"drug-response-prediction"
],
"tasks": [
"Drug Discovery",
"Drug Response Prediction"
],
"modalities": [
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "drugban",
"name": "DrugBAN",
"type": "model",
"url": "https://github.com/peizhenbai/DrugBAN",
"description": "Bilinear attention network for interpretable DTI prediction.",
"tags": [
"drug-discovery",
"drug-target-interaction"
],
"tasks": [
"Drug Discovery",
"Drug Target Interaction"
],
"modalities": [
"Protein",
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "druml",
"name": "DRUML",
"type": "model",
"url": "https://github.com/CutillasLab/DRUMLR",
"description": "Ensemble machine learning framework combining standard ML with deep learning to systematically rank anti-cancer drugs from proteomics and RNA-seq data.",
"tags": [
"drug-discovery",
"drug-response-prediction"
],
"tasks": [
"Drug Discovery",
"Drug Response Prediction"
],
"modalities": [
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "dtinet",
"name": "DTINet",
"type": "model",
"url": "https://github.com/luoyunan/DTINet",
"description": "Network-based framework integrating heterogeneous biological data for DTI prediction.",
"tags": [
"drug-discovery",
"drug-target-interaction"
],
"tasks": [
"Drug Discovery",
"Drug Target Interaction"
],
"modalities": [
"Protein",
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "enformer",
"name": "Enformer",
"type": "model",
"url": "https://github.com/deepmind/deepmind-research/tree/master/enformer",
"description": "Transformer model predicting gene expression from DNA sequence.",
"tags": [
"foundation-models",
"genomics-foundation-models"
],
"tasks": [
"Foundation Model"
],
"modalities": [
"Genomics"
],
"organism": [],
"api": false
},
{
"id": "esm3",
"name": "ESM3",
"type": "model",
"url": "https://github.com/evolutionaryscale/esm",
"description": "Multimodal protein language model that jointly reasons over sequence, structure, and function for generative protein design and engineering.",
"tags": [
"foundation-models",
"protein-foundation-models",
"protein-structure-prediction-and-design"
],
"tasks": [
"protein-sequence-design",
"representation-learning"
],
"modalities": [
"molecular-structure",
"protein-sequence"
],
"organism": [],
"api": false,
"entities": [
"protein"
],
"methods": [
"generative-model",
"language-model",
"transformer"
],
"github": "https://github.com/evolutionaryscale/esm",
"last_checked": "2026-08-08",
"metadata_sources": [
"https://github.com/evolutionaryscale/esm"
]
},
{
"id": "esmfold",
"name": "ESMFold",
"type": "model",
"url": "https://github.com/facebookresearch/esm",
"description": "Fast protein structure prediction using language model embeddings.",
"tags": [
"foundation-models",
"protein-foundation-models",
"protein-structure-prediction-and-design"
],
"tasks": [
"representation-learning",
"structure-prediction"
],
"modalities": [
"molecular-structure",
"protein-sequence"
],
"organism": [],
"api": false,
"entities": [
"protein"
],
"methods": [
"language-model",
"transformer"
],
"year": 2023,
"github": "https://github.com/facebookresearch/esm",
"last_checked": "2026-08-08",
"metadata_sources": [
"https://github.com/facebookresearch/esm"
]
},
{
"id": "evo",
"name": "Evo",
"type": "model",
"url": "https://github.com/evo-design/evo",
"description": "Long-context genomic foundation model (up to 1M tokens).",
"tags": [
"foundation-models",
"genomics-foundation-models"
],
"tasks": [
"Foundation Model"
],
"modalities": [
"Genomics"
],
"organism": [],
"api": false
},
{
"id": "evodiff",
"name": "EvoDiff",
"type": "model",
"url": "https://github.com/microsoft/evodiff",
"description": "Discrete diffusion framework for protein sequence generation trained on evolutionary-scale data, supporting unconditional generation, disordered region design, and functional motif scaffolding. [ [paper-2023](https://www.biorxiv.org/content/10.1101/2023.09.11.556673v1) ]",
"tags": [
"foundation-models",
"protein-foundation-models",
"protein-structure-prediction-and-design"
],
"tasks": [
"Foundation Model",
"Protein Structure Prediction"
],
"modalities": [
"Protein"
],
"organism": [],
"api": false
},
{
"id": "evolutionary_scale_modeling_esm",
"name": "Evolutionary Scale Modeling (ESM)",
"type": "model",
"url": "https://github.com/facebookresearch/esm",
"description": "Protein embeddings.",
"tags": [
"foundation-models",
"pre-trained-embedding",
"protein-foundation-models"
],
"tasks": [
"representation-learning"
],
"modalities": [
"protein-sequence"
],
"organism": [],
"api": false,
"entities": [
"protein"
],
"methods": [
"language-model",
"self-supervised-learning",
"transformer"
],
"github": "https://github.com/facebookresearch/esm",
"last_checked": "2026-08-08",
"metadata_sources": [
"https://github.com/facebookresearch/esm"
]
},
{
"id": "gears",
"name": "GEARS",
"type": "model",
"url": "https://github.com/snap-stanford/GEARS",
"description": "Graph-based model for predicting transcriptional responses to single and combinatorial genetic perturbations using biological priors.",
"tags": [
"foundation-models",
"single-cell-foundation-models",
"transcriptomics-foundation-models"
],
"tasks": [
"Foundation Model"
],
"modalities": [
"Single Cell",
"Transcriptomics"
],
"organism": [],
"api": false
},
{
"id": "genecompass",
"name": "GeneCompass",
"type": "model",
"url": "https://github.com/xCompass-AI/GeneCompass",
"description": "Large-scale foundation model integrating DNA regulatory sequences and single-cell transcriptomics from 120M+ cells across multiple species for gene regulation prediction.",
"tags": [
"foundation-models",
"multi-omics-foundation-models",
"single-cell-foundation-models"
],
"tasks": [
"foundation-model-pretraining",
"representation-learning"
],
"modalities": [
"single-cell-rna-seq",
"transcriptomics"
],
"organism": [],
"api": false,
"entities": [
"cell",
"gene"
],
"methods": [
"self-supervised-learning",
"transformer"
],
"year": 2024,
"github": "https://github.com/xCompass-AI/GeneCompass",
"paper": "https://www.nature.com/articles/s41422-024-01034-y",
"last_checked": "2026-08-08",
"metadata_sources": [
"https://github.com/xCompass-AI/GeneCompass",
"https://www.nature.com/articles/s41422-024-01034-y"
]
},
{
"id": "geneformer",
"name": "Geneformer",
"type": "model",
"url": "https://huggingface.co/ctheodoris/Geneformer",
"description": "Context-aware, attention-based deep learning model pretrained on a large corpus of single-cell transcriptomes.",
"tags": [
"foundation-models",
"single-cell-foundation-models",
"transcriptomics-foundation-models"
],
"tasks": [
"classification",
"foundation-model-pretraining",
"perturbation-prediction",
"representation-learning"
],
"modalities": [
"single-cell-rna-seq",
"transcriptomics"
],
"organism": [],
"api": false,
"entities": [
"cell",
"gene"
],
"methods": [
"self-supervised-learning",
"transformer"
],
"year": 2023,
"documentation": "https://geneformer.readthedocs.io/",
"paper": "https://www.nature.com/articles/s41586-023-06139-9",
"last_checked": "2026-08-08",
"metadata_sources": [
"https://huggingface.co/ctheodoris/Geneformer",
"https://www.nature.com/articles/s41586-023-06139-9"
]
},
{
"id": "genegpt",
"name": "GeneGPT",
"type": "model",
"url": "https://github.com/ncbi/GeneGPT",
"description": "LLM for biomedical information, integrated with various APIs.",
"tags": [
"llm-for-biology"
],
"tasks": [
"Language Modeling"
],
"modalities": [
"Text"
],
"organism": [],
"api": false
},
{
"id": "genept",
"name": "GenePT",
"type": "model",
"url": "https://github.com/yiqunchen/GenePT",
"description": "Foundation LLM for single-cell data.",
"tags": [
"llm-for-biology"
],
"tasks": [
"batch-correction",
"classification",
"representation-learning"
],
"modalities": [
"single-cell-rna-seq",
"transcriptomics"
],
"organism": [],
"api": false,
"entities": [
"cell",
"gene"
],
"methods": [
"language-model"
],
"year": 2023,
"github": "https://github.com/yiqunchen/GenePT",
"paper": "https://www.biorxiv.org/content/10.1101/2023.10.16.562533v2",
"last_checked": "2026-08-08",
"metadata_sources": [
"https://github.com/yiqunchen/GenePT",
"https://www.biorxiv.org/content/10.1101/2023.10.16.562533v2"
]
},
{
"id": "gigapath",
"name": "GigaPath",
"type": "model",
"url": "https://github.com/prov-gigapath/prov-gigapath",
"description": "Slide-level digital pathology foundation model pretrained on 1.3 billion pathology image tokens from whole-slide images.",
"tags": [
"foundation-models",
"single-cell-foundation-models",
"spatial-foundation-models"
],
"tasks": [
"foundation-model-pretraining",
"representation-learning"
],
"modalities": [
"histopathology",
"imaging"
],
"organism": [],
"api": false,
"entities": [
"tissue"
],
"methods": [
"self-supervised-learning",
"transformer"
],
"github": "https://github.com/prov-gigapath/prov-gigapath",
"last_checked": "2026-08-08",
"metadata_sources": [
"https://github.com/prov-gigapath/prov-gigapath"
]
},
{
"id": "glue",
"name": "GLUE",
"type": "model",
"url": "https://github.com/gao-lab/GLUE",
"description": "Graph-Linked Unified Embedding framework for unpaired single-cell multi-omics data integration across RNA, ATAC, methylation, and protein modalities.",
"tags": [
"foundation-models",
"multi-omics-foundation-models",
"single-cell-foundation-models"
],
"tasks": [
"Foundation Model"
],
"modalities": [
"Multi-Omics",
"Single Cell"
],
"organism": [],
"api": false
},
{
"id": "gpn_genomic_pre_trained_network",
"name": "GPN (Genomic Pre-trained Network)",
"type": "model",
"url": "https://github.com/songlab-cal/gpn",
"description": "Masked language model for DNA sequences enabling zero-shot variant effect prediction without requiring functional annotations.",
"tags": [
"foundation-models",
"genomics-foundation-models"
],
"tasks": [
"Foundation Model"
],
"modalities": [
"Genomics"
],
"organism": [],
"api": false
},
{
"id": "graphdta",
"name": "GraphDTA",
"type": "model",
"url": "https://github.com/thinng/GraphDTA",
"description": "Graph neural networkโ€“based DTI prediction using molecular graphs.",
"tags": [
"drug-discovery",
"drug-target-interaction"
],
"tasks": [
"Drug Discovery",
"Drug Target Interaction"
],
"modalities": [
"Protein",
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "grover",
"name": "GROVER",
"type": "model",
"url": "https://github.com/tencent-ailab/grover",
"description": "Self-supervised graph transformer for large-scale molecular representation learning from unlabeled compounds.",
"tags": [
"compound-embedding",
"compound-foundation-models",
"foundation-models"
],
"tasks": [
"representation-learning"
],
"modalities": [
"chemical-structure"
],
"organism": [],
"api": false,
"entities": [
"molecule"
],
"methods": [
"graph-neural-network",
"self-supervised-learning",
"transformer"
],
"github": "https://github.com/tencent-ailab/grover",
"last_checked": "2026-08-08",
"metadata_sources": [
"https://github.com/tencent-ailab/grover"
]
},
{
"id": "hidra",
"name": "HiDRA",
"type": "model",
"url": "https://github.com/bsml320/HiDRA",
"description": "Hierarchical network model incorporating gene and pathway-level information for cancer drug response prediction.",
"tags": [
"drug-discovery",
"drug-response-prediction"
],
"tasks": [
"Drug Discovery",
"Drug Response Prediction"
],
"modalities": [
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "hyenadna",
"name": "HyenaDNA",
"type": "model",
"url": "https://github.com/HazyResearch/hyena-dna",
"description": "Long-range genomic foundation model handling sequences up to 1M tokens with sub-quadratic attention.",
"tags": [
"foundation-models",
"genomics-foundation-models"
],
"tasks": [
"Foundation Model"
],
"modalities": [
"Genomics"
],
"organism": [],
"api": false
},
{
"id": "jamie",
"name": "JAMIE",
"type": "model",
"url": "https://github.com/Oafish1/JAMIE",
"description": "Joint variational autoencoder for multimodal single-cell data imputation and embedding.",
"tags": [
"foundation-models",
"multi-omics-foundation-models",
"single-cell-foundation-models"
],
"tasks": [
"Foundation Model"
],
"modalities": [
"Multi-Omics",
"Single Cell"
],
"organism": [],
"api": false
},
{
"id": "jtvae",
"name": "JTVAE",
"type": "model",
"url": "https://github.com/wengong-jin/icml18-jtnn",
"description": "Junction tree variational autoencoder for molecular graph generation that guarantees chemical validity via a hierarchical tree decomposition.",
"tags": [
"drug-discovery",
"molecular-generation"
],
"tasks": [
"Drug Discovery",
"Molecular Generation"
],
"modalities": [
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "matcha",
"name": "Matcha",
"type": "model",
"url": "https://github.com/LigandPro/Matcha",
"description": "Multi-stage Riemannian flow matching model for physically valid molecular docking with scoring, pose filtering, and benchmarks.",
"tags": [
"drug-discovery",
"molecular-generation"
],
"tasks": [
"Drug Discovery",
"Molecular Generation"
],
"modalities": [
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "mcpinn",
"name": "MCPINN",
"type": "model",
"url": "https://github.com/mhlee0903/multi_channels_PINN",
"description": "Drug discovery via compound-protein interaction and machine learning.",
"tags": [
"compound-protein-interaction",
"drug-discovery"
],
"tasks": [
"Compound-Protein Interaction",
"Drug Discovery"
],
"modalities": [
"Protein",
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "midas",
"name": "MIDAS",
"type": "model",
"url": "https://github.com/labomics/midas",
"description": "Mosaic integration and differential accessibility model for single-cell multi-omics data that handles arbitrary missing-modality combinations across transcriptomics, chromatin accessibility, and proteomics.",
"tags": [
"foundation-models",
"multi-omics-foundation-models",
"single-cell-foundation-models"
],
"tasks": [
"Foundation Model"
],
"modalities": [
"Multi-Omics",
"Single Cell"
],
"organism": [],
"api": false
},
{
"id": "mira",
"name": "MIRA",
"type": "model",
"url": "https://github.com/cistrome/MIRA",
"description": "Probabilistic multimodal topic model jointly modeling single-cell transcriptomics and chromatin accessibility for regulatory network inference.",
"tags": [
"foundation-models",
"multi-omics-foundation-models",
"single-cell-foundation-models"
],
"tasks": [
"Foundation Model"
],
"modalities": [
"Multi-Omics",
"Single Cell"
],
"organism": [],
"api": false
},
{
"id": "mofa",
"name": "MOFA+",
"type": "model",
"url": "https://github.com/bioFAM/MOFA2",
"description": "Multi-Omics Factor Analysis framework identifying shared axes of variation across bulk and single-cell datasets including RNA, ATAC, proteomics, methylation, and copy number.",
"tags": [
"foundation-models",
"multi-omics-foundation-models",
"single-cell-foundation-models"
],
"tasks": [
"Foundation Model"
],
"modalities": [
"Multi-Omics",
"Single Cell"
],
"organism": [],
"api": false
},
{
"id": "mofgcn",
"name": "MOFGCN",
"type": "model",
"url": "https://github.com/weiba/MOFGCN/tree/main",
"description": "GCN + heterogeneous network.",
"tags": [
"drug-discovery",
"drug-response-prediction"
],
"tasks": [
"Drug Discovery",
"Drug Response Prediction"
],
"modalities": [
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "mol2vec",
"name": "Mol2Vec",
"type": "model",
"url": "https://github.com/samoturk/mol2vec",
"description": "Unsupervised molecular embedding method inspired by Word2Vec for learning vector representations of chemical substructures.",
"tags": [
"compound-embedding",
"compound-foundation-models",
"foundation-models"
],
"tasks": [
"Foundation Model"
],
"modalities": [
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "molecular_transformer",
"name": "Molecular Transformer",
"type": "model",
"url": "https://github.com/pschwllr/MolecularTransformer",
"description": "Sequence-to-sequence model for retrosynthesis prediction.",
"tags": [
"drug-discovery",
"molecular-generation"
],
"tasks": [
"Drug Discovery",
"Molecular Generation"
],
"modalities": [
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "molformer",
"name": "MolFormer",
"type": "model",
"url": "https://github.com/IBM/molformer",
"description": "Linear attention transformer pretrained on millions of SMILES strings for efficient molecular embeddings.",
"tags": [
"compound-embedding",
"compound-foundation-models",
"foundation-models"
],
"tasks": [
"representation-learning"
],
"modalities": [
"chemical-structure"
],
"organism": [],
"api": false,
"entities": [
"molecule"
],
"methods": [
"language-model",
"self-supervised-learning",
"transformer"
],
"github": "https://github.com/IBM/molformer",
"last_checked": "2026-08-08",
"metadata_sources": [
"https://github.com/IBM/molformer"
]
},
{
"id": "molgpt",
"name": "MolGPT",
"type": "model",
"url": "https://github.com/devalab/molgpt",
"description": "Transformer-based model for molecular generation.",
"tags": [
"drug-discovery",
"molecular-generation"
],
"tasks": [
"Drug Discovery",
"Molecular Generation"
],
"modalities": [
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "molt5",
"name": "MolT5",
"type": "model",
"url": "https://github.com/blender-nlp/MolT5",
"description": "Language model for molecular tasks bridging text and SMILES, enabling molecule captioning and text-driven molecule generation.",
"tags": [
"llm-for-biology"
],
"tasks": [
"Language Modeling"
],
"modalities": [
"Text"
],
"organism": [],
"api": false
},
{
"id": "moltrans",
"name": "MolTrans",
"type": "model",
"url": "https://github.com/kexinhuang12345/MolTrans",
"description": "Transformer-based DTI model leveraging molecular substructures.",
"tags": [
"drug-discovery",
"drug-target-interaction"
],
"tasks": [
"Drug Discovery",
"Drug Target Interaction"
],
"modalities": [
"Protein",
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "multigrate",
"name": "Multigrate",
"type": "model",
"url": "https://github.com/theislab/multigrate",
"description": "Asymmetric multi-omics variational autoencoder for integrating single-cell data across RNA, ATAC, and protein modalities with missing-modality support.",
"tags": [
"foundation-models",
"multi-omics-foundation-models",
"single-cell-foundation-models"
],
"tasks": [
"Foundation Model"
],
"modalities": [
"Multi-Omics",
"Single Cell"
],
"organism": [],
"api": false
},
{
"id": "multivi",
"name": "MultiVI",
"type": "model",
"url": "https://github.com/scverse/scvi-tools",
"description": "Multi-modal variational autoencoder for integrating paired and unpaired single-cell RNA-seq and ATAC-seq measurements into a unified latent space.",
"tags": [
"foundation-models",
"multi-omics-foundation-models",
"single-cell-foundation-models"
],
"tasks": [
"Foundation Model"
],
"modalities": [
"Multi-Omics",
"Single Cell"
],
"organism": [],
"api": false
},
{
"id": "musk",
"name": "MUSK",
"type": "model",
"url": "https://github.com/lilab-stanford/MUSK",
"description": "Vision-language foundation model for precision oncology analyzing multimodal paired text and pathology image data for biomarker prediction and retrieval.",
"tags": [
"foundation-models",
"multi-modal-foundation-models"
],
"tasks": [
"Foundation Model"
],
"modalities": [
"Multi-Modal"
],
"organism": [],
"api": false
},
{
"id": "neodti",
"name": "NeoDTI",
"type": "model",
"url": "https://github.com/FangpingWan/NeoDTI",
"description": "Library for drug-target interaction prediction.",
"tags": [
"drug-discovery",
"drug-target-interaction"
],
"tasks": [
"Drug Discovery",
"Drug Target Interaction"
],
"modalities": [
"Protein",
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "nicheformer",
"name": "Nicheformer",
"type": "model",
"url": "https://github.com/theislab/nicheformer",
"description": "Foundation model for single-cell and spatial omics using a transformer architecture with positional embeddings to encode spatial cell information.",
"tags": [
"foundation-models",
"single-cell-foundation-models",
"spatial-foundation-models"
],
"tasks": [
"foundation-model-pretraining",
"representation-learning"
],
"modalities": [
"single-cell-rna-seq",
"spatial-transcriptomics",
"transcriptomics"
],
"organism": [],
"api": false,
"entities": [
"cell",
"gene",
"tissue"
],
"methods": [
"self-supervised-learning",
"transformer"
],
"year": 2024,
"github": "https://github.com/theislab/nicheformer",
"paper": "https://doi.org/10.1101/2024.04.15.589472",
"last_checked": "2026-08-08",
"metadata_sources": [
"https://github.com/theislab/nicheformer",
"https://doi.org/10.1101/2024.04.15.589472"
]
},
{
"id": "nucleotide_transformer",
"name": "Nucleotide Transformer",
"type": "model",
"url": "https://github.com/instadeepai/nucleotide-transformer",
"description": "Foundation model for genomic sequences across multiple species.",
"tags": [
"foundation-models",
"genomics-foundation-models"
],
"tasks": [
"Foundation Model"
],
"modalities": [
"Genomics"
],
"organism": [],
"api": false
},
{
"id": "omegafold",
"name": "OmegaFold",
"type": "model",
"url": "https://github.com/HeliXonProtein/OmegaFold",
"description": "High-resolution de novo protein structure prediction from sequence.",
"tags": [
"foundation-models",
"protein-foundation-models",
"protein-structure-prediction-and-design"
],
"tasks": [
"Foundation Model",
"Protein Structure Prediction"
],
"modalities": [
"Protein"
],
"organism": [],
"api": false
},
{
"id": "openfold",
"name": "OpenFold",
"type": "model",
"url": "https://github.com/aqlaboratory/openfold",
"description": "Trainable, memory-efficient open-source reproduction of AlphaFold2 enabling custom protein structure prediction workflows.",
"tags": [
"foundation-models",
"protein-foundation-models",
"protein-structure-prediction-and-design"
],
"tasks": [
"Foundation Model",
"Protein Structure Prediction"
],
"modalities": [
"Protein"
],
"organism": [],
"api": false
},
{
"id": "paccmannrl",
"name": "PaccMannRL",
"type": "model",
"url": "https://github.com/PaccMann/paccmann_generator",
"description": "Reinforcement learning-based generative model for de novo hit-like anticancer molecule design from transcriptomic data.",
"tags": [
"drug-discovery",
"molecular-generation"
],
"tasks": [
"Drug Discovery",
"Molecular Generation"
],
"modalities": [
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "pathomicfusion",
"name": "PathomicFusion",
"type": "model",
"url": "https://github.com/mahmoodlab/PathomicFusion",
"description": "Integrated framework fusing histopathology and genomic features via CNN, GNN, and attention gating for cancer diagnosis and prognosis.",
"tags": [
"foundation-models",
"multi-modal-foundation-models"
],
"tasks": [
"Foundation Model"
],
"modalities": [
"Multi-Modal"
],
"organism": [],
"api": false
},
{
"id": "phikon",
"name": "Phikon",
"type": "model",
"url": "https://huggingface.co/owkin/phikon",
"description": "ViT-based pathology foundation model pretrained with iBOT self-supervision on TCGA whole-slide images.",
"tags": [
"foundation-models",
"single-cell-foundation-models",
"spatial-foundation-models"
],
"tasks": [
"foundation-model-pretraining",
"representation-learning"
],
"modalities": [
"histopathology",
"imaging"
],
"organism": [],
"api": false,
"entities": [
"tissue"
],
"methods": [
"self-supervised-learning",
"transformer"
],
"documentation": "https://huggingface.co/owkin/phikon",
"last_checked": "2026-08-08",
"metadata_sources": [
"https://huggingface.co/owkin/phikon"
]
},
{
"id": "plip",
"name": "PLIP",
"type": "model",
"url": "https://github.com/PathologyFoundation/plip",
"description": "Vision-language foundation model for pathology trained with contrastive learning on pathology imageโ€“text pairs for image classification and text-to-image retrieval.",
"tags": [
"foundation-models",
"multi-modal-foundation-models"
],
"tasks": [
"classification",
"representation-learning"
],
"modalities": [
"histopathology",
"imaging"
],
"organism": [],
"api": false,
"entities": [
"tissue"
],
"methods": [
"contrastive-learning"
],
"github": "https://github.com/PathologyFoundation/plip",
"last_checked": "2026-08-08",
"metadata_sources": [
"https://github.com/PathologyFoundation/plip"
]
},
{
"id": "porpoise",
"name": "PORPOISE",
"type": "model",
"url": "https://github.com/mahmoodlab/PORPOISE",
"description": "Pan-cancer integrative histology-genomic analysis framework using multimodal deep learning for patient stratification.",
"tags": [
"foundation-models",
"multi-modal-foundation-models"
],
"tasks": [
"Foundation Model"
],
"modalities": [
"Multi-Modal"
],
"organism": [],
"api": false
},
{
"id": "prnet",
"name": "PRNet",
"type": "model",
"url": "https://github.com/Perturbation-Response-Prediction/PRnet",
"description": "Deep generative model for predicting transcriptional responses to novel chemical perturbations for drug discovery.",
"tags": [
"drug-discovery",
"drug-perturbation"
],
"tasks": [
"Drug Discovery",
"Drug Perturbation"
],
"modalities": [
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "progen2",
"name": "ProGen2",
"type": "model",
"url": "https://github.com/salesforce/progen",
"description": "Protein language model trained on diverse protein families for sequence generation and fitness prediction.",
"tags": [
"foundation-models",
"pre-trained-embedding",
"protein-foundation-models"
],
"tasks": [
"protein-sequence-design",
"representation-learning"
],
"modalities": [
"protein-sequence"
],
"organism": [],
"api": false,
"entities": [
"protein"
],
"methods": [
"generative-model",
"language-model",
"transformer"
],
"github": "https://github.com/salesforce/progen",
"last_checked": "2026-08-08",
"metadata_sources": [
"https://github.com/salesforce/progen"
]
},
{
"id": "proteinmpnn",
"name": "ProteinMPNN",
"type": "model",
"url": "https://github.com/dauparas/ProteinMPNN",
"description": "Deep learning model for protein sequence design given backbone structure.",
"tags": [
"foundation-models",
"protein-foundation-models",
"protein-structure-prediction-and-design"
],
"tasks": [
"protein-sequence-design"
],
"modalities": [
"molecular-structure",
"protein-sequence"
],
"organism": [],
"api": false,
"entities": [
"protein"
],
"methods": [
"graph-neural-network",
"message-passing-neural-network"
],
"year": 2022,
"github": "https://github.com/dauparas/ProteinMPNN",
"last_checked": "2026-08-08",
"metadata_sources": [
"https://github.com/dauparas/ProteinMPNN"
]
},
{
"id": "prottrans",
"name": "ProtTrans",
"type": "model",
"url": "https://github.com/agemagician/ProtTrans",
"description": "Suite of protein language models (ProtBERT, ProtT5, ProtXLNet) trained on billions of protein sequences from UniRef and BFD.",
"tags": [
"foundation-models",
"pre-trained-embedding",
"protein-foundation-models"
],
"tasks": [
"representation-learning"
],
"modalities": [
"protein-sequence"
],
"organism": [],
"api": false,
"entities": [
"protein"
],
"methods": [
"language-model",
"self-supervised-learning",
"transformer"
],
"github": "https://github.com/agemagician/ProtTrans",
"last_checked": "2026-08-08",
"metadata_sources": [
"https://github.com/agemagician/ProtTrans"
]
},
{
"id": "recover",
"name": "RECOVER",
"type": "model",
"url": "https://github.com/RECOVERcoalition/Recover",
"description": "Machine learning framework for predicting synergistic drug combination responses across cell lines.",
"tags": [
"drug-discovery",
"drug-response-prediction"
],
"tasks": [
"Drug Discovery",
"Drug Response Prediction"
],
"modalities": [
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "reinvent",
"name": "REINVENT",
"type": "model",
"url": "https://github.com/MolecularAI/Reinvent",
"description": "Reinforcement learning for de novo drug design.",
"tags": [
"drug-discovery",
"molecular-generation"
],
"tasks": [
"Drug Discovery",
"Molecular Generation"
],
"modalities": [
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "release",
"name": "ReLeaSE",
"type": "model",
"url": "https://github.com/isayev/ReLeaSE",
"description": "Deep reinforcement learning framework for de novo drug design combining a generative and predictive model.",
"tags": [
"drug-discovery",
"molecular-generation"
],
"tasks": [
"Drug Discovery",
"Molecular Generation"
],
"modalities": [
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "rfdiffusion",
"name": "RFdiffusion",
"type": "model",
"url": "https://github.com/RosettaCommons/RFdiffusion",
"description": "Generative model for protein backbone design using diffusion.",
"tags": [
"foundation-models",
"protein-foundation-models",
"protein-structure-prediction-and-design"
],
"tasks": [
"Foundation Model",
"Protein Structure Prediction"
],
"modalities": [
"Protein"
],
"organism": [],
"api": false
},
{
"id": "rosettafold",
"name": "RoseTTAFold",
"type": "model",
"url": "https://github.com/RosettaCommons/RoseTTAFold",
"description": "Three-track neural network for protein structure prediction.",
"tags": [
"foundation-models",
"protein-foundation-models",
"protein-structure-prediction-and-design"
],
"tasks": [
"Foundation Model",
"Protein Structure Prediction"
],
"modalities": [
"Protein"
],
"organism": [],
"api": false
},
{
"id": "saprot",
"name": "SaProt",
"type": "model",
"url": "https://github.com/westlake-reup/SaProt",
"description": "Structure-aware protein language model using structure-aware tokens that encode both sequence and backbone geometry for improved function prediction.",
"tags": [
"foundation-models",
"protein-foundation-models",
"protein-structure-prediction-and-design"
],
"tasks": [
"Foundation Model",
"Protein Structure Prediction"
],
"modalities": [
"Protein"
],
"organism": [],
"api": false
},
{
"id": "saturn",
"name": "SATURN",
"type": "model",
"url": "https://github.com/snap-stanford/SATURN",
"description": "Transformer-based model integrating gene expression and protein sequences via a protein language model to learn unified multi-species cell embeddings.",
"tags": [
"foundation-models",
"single-cell-foundation-models",
"transcriptomics-foundation-models"
],
"tasks": [
"Foundation Model"
],
"modalities": [
"Single Cell",
"Transcriptomics"
],
"organism": [],
"api": false
},
{
"id": "scarches",
"name": "scArches",
"type": "model",
"url": "https://github.com/theislab/scarches",
"description": "Transfer learning framework for mapping new single-cell datasets onto pre-trained reference atlases across batches, conditions, and modalities.",
"tags": [
"domain-alignment",
"foundation-models",
"single-cell-foundation-models"
],
"tasks": [
"Domain Alignment",
"Foundation Model"
],
"modalities": [
"Single Cell"
],
"organism": [],
"api": false
},
{
"id": "scbert",
"name": "scBERT",
"type": "model",
"url": "https://github.com/TencentAILabHealthcare/scBERT",
"description": "BERT-based foundation model pretrained on large-scale scRNA-seq data for cell type annotation.",
"tags": [
"foundation-models",
"single-cell-foundation-models",
"transcriptomics-foundation-models"
],
"tasks": [
"cell-type-annotation",
"classification",
"foundation-model-pretraining"
],
"modalities": [
"single-cell-rna-seq",
"transcriptomics"
],
"organism": [],
"api": false,
"entities": [
"cell",
"gene"
],
"methods": [
"language-model",
"self-supervised-learning",
"transformer"
],
"year": 2022,
"github": "https://github.com/TencentAILabHealthcare/scBERT",
"paper": "https://www.nature.com/articles/s42256-022-00534-z",
"last_checked": "2026-08-08",
"metadata_sources": [
"https://github.com/TencentAILabHealthcare/scBERT",
"https://www.nature.com/articles/s42256-022-00534-z"
]
},
{
"id": "scbutterfly",
"name": "scButterfly",
"type": "model",
"url": "https://github.com/BioX-NKU/scButterfly",
"description": "Dual-aligned variational autoencoder for single-cell cross-modality translation between paired and unpaired multiomics data.",
"tags": [
"foundation-models",
"multi-omics-foundation-models",
"single-cell-foundation-models"
],
"tasks": [
"Foundation Model"
],
"modalities": [
"Multi-Omics",
"Single Cell"
],
"organism": [],
"api": false
},
{
"id": "scfoundation",
"name": "scFoundation",
"type": "model",
"url": "https://github.com/biomap-research/scFoundation",
"description": "Large-scale foundation model for single-cell gene expression, enabling multiple downstream tasks.",
"tags": [
"foundation-models",
"single-cell-foundation-models",
"transcriptomics-foundation-models"
],
"tasks": [
"cell-type-annotation",
"drug-response-prediction",
"foundation-model-pretraining",
"perturbation-prediction",
"representation-learning"
],
"modalities": [
"single-cell-rna-seq",
"transcriptomics"
],
"organism": [],
"api": false,
"entities": [
"cell",
"gene"
],
"methods": [
"self-supervised-learning",
"transformer"
],
"year": 2024,
"github": "https://github.com/biomap-research/scFoundation",
"paper": "https://www.nature.com/articles/s41592-024-02305-7",
"last_checked": "2026-08-08",
"metadata_sources": [
"https://github.com/biomap-research/scFoundation",
"https://www.nature.com/articles/s41592-024-02305-7"
]
},
{
"id": "scgpt",
"name": "scGPT",
"type": "model",
"url": "https://github.com/bowang-lab/scGPT",
"description": "Transformer-based foundation model pretrained on millions of single-cell profiles.",
"tags": [
"foundation-models",
"single-cell-foundation-models",
"transcriptomics-foundation-models"
],
"tasks": [
"cell-type-annotation",
"foundation-model-pretraining",
"gene-regulatory-network-inference",
"perturbation-prediction",
"representation-learning"
],
"modalities": [
"multi-omics",
"single-cell-rna-seq",
"transcriptomics"
],
"organism": [],
"api": false,
"entities": [
"cell",
"gene"
],
"methods": [
"generative-model",
"self-supervised-learning",
"transformer"
],
"year": 2024,
"github": "https://github.com/bowang-lab/scGPT",
"documentation": "https://scgpt.readthedocs.io/en/latest/",
"paper": "https://www.nature.com/articles/s41592-024-02201-0",
"last_checked": "2026-08-08",
"metadata_sources": [
"https://github.com/bowang-lab/scGPT",
"https://www.nature.com/articles/s41592-024-02201-0"
]
},
{
"id": "scgpt_spatial",
"name": "scGPT-spatial",
"type": "model",
"url": "https://github.com/bowang-lab/scGPT-spatial",
"description": "Extension of scGPT for spatial transcriptomics with continual pretraining and a mixture-of-experts decoder for spatial gene expression analysis.",
"tags": [
"foundation-models",
"single-cell-foundation-models",
"spatial-foundation-models"
],
"tasks": [
"foundation-model-pretraining",
"imputation",
"representation-learning"
],
"modalities": [
"multi-omics",
"single-cell-rna-seq",
"spatial-transcriptomics"
],
"organism": [],
"api": false,
"entities": [
"cell",
"gene",
"tissue"
],
"methods": [
"generative-model",
"self-supervised-learning",
"transformer"
],
"year": 2025,
"github": "https://github.com/bowang-lab/scGPT-spatial",
"paper": "https://www.biorxiv.org/content/10.1101/2025.02.05.636714v1",
"last_checked": "2026-08-08",
"metadata_sources": [
"https://github.com/bowang-lab/scGPT-spatial",
"https://www.biorxiv.org/content/10.1101/2025.02.05.636714v1"
]
},
{
"id": "scmulan",
"name": "scMulan",
"type": "model",
"url": "https://github.com/SuperBianC/scMulan",
"description": "Single-cell multi-omic language model pretrained on ~10M cells spanning transcriptomics, epigenomics, and proteomics for cross-omics transfer tasks.",
"tags": [
"foundation-models",
"multi-omics-foundation-models",
"single-cell-foundation-models"
],
"tasks": [
"foundation-model-pretraining",
"representation-learning"
],
"modalities": [
"epigenomics",
"multi-omics",
"proteomics",
"single-cell-rna-seq",
"transcriptomics"
],
"organism": [],
"api": false,
"entities": [
"cell",
"gene"
],
"methods": [
"language-model",
"transformer"
],
"github": "https://github.com/SuperBianC/scMulan",
"last_checked": "2026-08-08",
"metadata_sources": [
"https://github.com/SuperBianC/scMulan"
]
},
{
"id": "scpair",
"name": "scPair",
"type": "model",
"url": "https://github.com/quon-titative-biology/scPair",
"description": "Bidirectional feedforward network for single-cell multimodal analysis with cross-modality prediction leveraging single-cell atlases.",
"tags": [
"foundation-models",
"multi-omics-foundation-models",
"single-cell-foundation-models"
],
"tasks": [
"Foundation Model"
],
"modalities": [
"Multi-Omics",
"Single Cell"
],
"organism": [],
"api": false
},
{
"id": "scprint",
"name": "scPRINT",
"type": "model",
"url": "https://github.com/cantinilab/scPRINT",
"description": "Pretrained on 50M cells for scRNA-seq denoising & zero imputation.",
"tags": [
"llm-for-biology"
],
"tasks": [
"batch-correction",
"cell-type-annotation",
"foundation-model-pretraining",
"gene-regulatory-network-inference",
"imputation",
"representation-learning"
],
"modalities": [
"single-cell-rna-seq",
"transcriptomics"
],
"organism": [],
"api": false,
"entities": [
"cell",
"gene"
],
"methods": [
"self-supervised-learning",
"transformer"
],
"year": 2025,
"github": "https://github.com/cantinilab/scPRINT",
"documentation": "https://www.jkobject.com/scPRINT/",
"paper": "https://www.nature.com/articles/s41467-025-58699-1",
"last_checked": "2026-08-08",
"metadata_sources": [
"https://github.com/cantinilab/scPRINT",
"https://www.nature.com/articles/s41467-025-58699-1"
]
},
{
"id": "sei",
"name": "Sei",
"type": "model",
"url": "https://github.com/FunctionLab/sei-framework",
"description": "Sequence-to-function framework learning a genome-wide regulatory activity code from DNA sequences for variant effect prediction.",
"tags": [
"foundation-models",
"genomics-foundation-models"
],
"tasks": [
"Foundation Model"
],
"modalities": [
"Genomics"
],
"organism": [],
"api": false
},
{
"id": "spatialglue",
"name": "SpatialGlue",
"type": "model",
"url": "https://github.com/zhanglabtools/SpatialGlue",
"description": "Graph attention network for spatial multi-omics integration jointly embedding spatial transcriptomics with chromatin accessibility or proteomics.",
"tags": [
"foundation-models",
"multi-omics-foundation-models",
"single-cell-foundation-models"
],
"tasks": [
"Foundation Model"
],
"modalities": [
"Multi-Omics",
"Single Cell"
],
"organism": [],
"api": false
},
{
"id": "targetdiff",
"name": "TargetDiff",
"type": "model",
"url": "https://github.com/guanjq/targetdiff",
"description": "3D equivariant diffusion model for structure-based drug design.",
"tags": [
"drug-discovery",
"molecular-generation"
],
"tasks": [
"Drug Discovery",
"Molecular Generation"
],
"modalities": [
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "tgsa",
"name": "TGSA",
"type": "model",
"url": "https://github.com/violet-sto/TGSA",
"description": "Tumor gene set and attention-based model leveraging biological pathway knowledge for drug response prediction.",
"tags": [
"drug-discovery",
"drug-response-prediction"
],
"tasks": [
"Drug Discovery",
"Drug Response Prediction"
],
"modalities": [
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "toad",
"name": "TOAD",
"type": "model",
"url": "https://github.com/mahmoodlab/TOAD",
"description": "Tumor Origin Assessment via Deep-learning; weakly-supervised multi-task model predicting cancer primary origin from H&E whole-slide images.",
"tags": [
"foundation-models",
"multi-modal-foundation-models"
],
"tasks": [
"Foundation Model"
],
"modalities": [
"Multi-Modal"
],
"organism": [],
"api": false
},
{
"id": "tosica",
"name": "TOSICA",
"type": "model",
"url": "https://github.com/JackieHanlaopo/TOSICA",
"description": "Transformer-based framework for one-stop interpretable cell-type annotation supporting cross-dataset and cross-species transfer.",
"tags": [
"domain-alignment",
"foundation-models",
"single-cell-foundation-models"
],
"tasks": [
"Domain Alignment",
"Foundation Model"
],
"modalities": [
"Single Cell"
],
"organism": [],
"api": false
},
{
"id": "totalvi",
"name": "totalVI",
"type": "model",
"url": "https://github.com/scverse/scvi-tools",
"description": "Probabilistic framework for joint analysis of paired scRNA-seq and protein (CITE-seq) data enabling multi-modal cell state representation across single-cell datasets.",
"tags": [
"foundation-models",
"multi-omics-foundation-models",
"single-cell-foundation-models"
],
"tasks": [
"Foundation Model"
],
"modalities": [
"Multi-Omics",
"Single Cell"
],
"organism": [],
"api": false
},
{
"id": "transformercpi",
"name": "TransformerCPI",
"type": "model",
"url": "https://github.com/lifanchen-simm/transformerCPI",
"description": "CPI prediction using Transformer.",
"tags": [
"compound-protein-interaction",
"drug-discovery"
],
"tasks": [
"Compound-Protein Interaction",
"Drug Discovery"
],
"modalities": [
"Protein",
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "transigen",
"name": "TranSiGen",
"type": "model",
"url": "https://github.com/myzhengSIMM/TranSiGen",
"description": "Dual-VAE architecture for ligand-based virtual screening, drug response prediction, and drug repurposing using chemical-induced transcriptional profiles.",
"tags": [
"drug-discovery",
"drug-repurposing"
],
"tasks": [
"Drug Discovery",
"Drug Repurposing"
],
"modalities": [
"Small Molecule"
],
"organism": [],
"api": false
},
{
"id": "uce",
"name": "UCE",
"type": "model",
"url": "https://github.com/snap-stanford/UCE",
"description": "Universal Cell Embeddings: zero-shot single-cell embedding model trained on 36M cells across species, tissues, and assays without fine-tuning.",
"tags": [
"foundation-models",
"single-cell-foundation-models",
"transcriptomics-foundation-models"
],
"tasks": [
"foundation-model-pretraining",
"representation-learning"
],
"modalities": [
"single-cell-rna-seq",
"transcriptomics"
],
"organism": [],
"api": false,
"entities": [
"cell"
],
"methods": [
"self-supervised-learning"
],
"year": 2026,
"github": "https://github.com/snap-stanford/UCE",
"paper": "https://www.nature.com/articles/s41586-026-10689-z",
"last_checked": "2026-08-08",
"metadata_sources": [
"https://github.com/snap-stanford/UCE",
"https://www.nature.com/articles/s41586-026-10689-z"
]
},
{
"id": "uni",
"name": "UNI",
"type": "model",
"url": "https://github.com/mahmoodlab/UNI",
"description": "General-purpose self-supervised pathology foundation model trained on 100K+ whole-slide images for diverse computational pathology tasks.",
"tags": [
"foundation-models",
"single-cell-foundation-models",
"spatial-foundation-models"
],
"tasks": [
"foundation-model-pretraining",
"representation-learning"
],
"modalities": [
"histopathology",
"imaging"
],
"organism": [],
"api": false,
"entities": [
"tissue"
],
"methods": [
"self-supervised-learning",
"transformer"
],
"github": "https://github.com/mahmoodlab/UNI",
"last_checked": "2026-08-08",
"metadata_sources": [
"https://github.com/mahmoodlab/UNI"
]
},
{
"id": "uni_mol",
"name": "Uni-Mol",
"type": "model",
"url": "https://github.com/deepmodeling/Uni-Mol",
"description": "3D molecular pretraining framework for universal representation learning on molecules and protein pockets.",
"tags": [
"compound-embedding",
"compound-foundation-models",
"foundation-models"
],
"tasks": [
"docking",
"representation-learning"
],
"modalities": [
"chemical-structure",
"molecular-structure"
],
"organism": [],
"api": false,
"entities": [
"molecule",
"protein"
],
"methods": [
"self-supervised-learning",
"transformer"
],
"year": 2023,
"github": "https://github.com/deepmodeling/Uni-Mol",
"paper": "https://openreview.net/forum?id=6K2RM6wVqKu",
"last_checked": "2026-08-08",
"metadata_sources": [
"https://github.com/deepmodeling/Uni-Mol",
"https://openreview.net/forum?id=6K2RM6wVqKu"
]
},
{
"id": "unitednet",
"name": "UnitedNet",
"type": "model",
"url": "https://github.com/LiuLab-Bioelectronics-Harvard/UnitedNet",
"description": "Interpretable multi-task deep neural network for single-cell multi-omics integration spanning transcriptomics, chromatin accessibility, and proteomics.",
"tags": [
"foundation-models",
"multi-omics-foundation-models",
"single-cell-foundation-models"
],
"tasks": [
"Foundation Model"
],
"modalities": [
"Multi-Omics",
"Single Cell"
],
"organism": [],
"api": false
},
{
"id": "virchow",
"name": "Virchow",
"type": "model",
"url": "https://huggingface.co/paige-ai/Virchow",
"description": "Million-slide digital pathology foundation model using a vision transformer and self-supervised distillation for tile-level pathology image representation.",
"tags": [
"foundation-models",
"multi-modal-foundation-models"
],
"tasks": [
"Foundation Model"
],
"modalities": [
"Multi-Modal"
],
"organism": [],
"api": false
},
{
"id": "autozyme",
"name": "AutoZyme",
"type": "toolkit",
"url": "https://github.com/ElliotXie/autozyme",
"description": "Autonomous agentic framework that speeds up bioinformatics software (e.g. Scanpy, Seurat) on CPUs while preserving the original results.",
"tags": [
"preprocessing-tools"
],
"tasks": [
"Preprocessing"
],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "biopython",
"name": "Biopython",
"type": "toolkit",
"url": "https://biopython.org/",
"description": "Collection of Python tools for biological computation including sequence analysis, structure parsing, and database access.",
"tags": [
"preprocessing-tools"
],
"tasks": [
"Preprocessing"
],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "casper",
"name": "CaSpER",
"type": "toolkit",
"url": "https://github.com/akdess/CaSpER",
"description": "CNV identification and visualization by integrative analysis of single-cell or bulk RNA-seq data.",
"tags": [
"preprocessing-tools"
],
"tasks": [
"Preprocessing"
],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "cellcharter",
"name": "CellCharter",
"type": "toolkit",
"url": "https://github.com/CSOgroup/cellcharter",
"description": "Identification and characterization of spatial cell niches from spatial transcriptomics using VAEs and Gaussian mixture models.",
"tags": [
"preprocessing-tools"
],
"tasks": [
"Preprocessing"
],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "cellchat",
"name": "CellChat",
"type": "toolkit",
"url": "https://github.com/sqjin/CellChat",
"description": "Inference and analysis of cell-cell communication ligand-receptor networks from single-cell transcriptomics data.",
"tags": [
"preprocessing-tools"
],
"tasks": [
"Preprocessing"
],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "celltypist",
"name": "CellTypist",
"type": "toolkit",
"url": "https://github.com/Teichlab/celltypist",
"description": "Automated cell type annotation for scRNA-seq.",
"tags": [
"preprocessing-tools"
],
"tasks": [
"Preprocessing"
],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "chatspatial",
"name": "ChatSpatial",
"type": "toolkit",
"url": "https://github.com/cafferychen777/ChatSpatial",
"description": "MCP server for spatial transcriptomics analysis via natural language.",
"tags": [
"preprocessing-tools"
],
"tasks": [
"Preprocessing"
],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "chemistry_development_kit",
"name": "Chemistry Development Kit",
"type": "toolkit",
"url": "https://github.com/cdk/cdk",
"description": "Cheminformatics software & machine learning tools.",
"tags": [
"preprocessing-tools"
],
"tasks": [
"Preprocessing"
],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "commot",
"name": "COMMOT",
"type": "toolkit",
"url": "https://github.com/zcang/COMMOT",
"description": "Optimal transport-based framework for screening cell-cell communication in spatial transcriptomics.",
"tags": [
"preprocessing-tools"
],
"tasks": [
"Preprocessing"
],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "deepchem",
"name": "DeepChem",
"type": "toolkit",
"url": "https://github.com/deepchem/deepchem",
"description": "Deep learning library for drug discovery, quantum chemistry, and materials science.",
"tags": [
"preprocessing-tools"
],
"tasks": [
"Preprocessing"
],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "deeptalk",
"name": "DeepTalk",
"type": "toolkit",
"url": "https://github.com/JiangBioLab/DeepTalk",
"description": "Graph attention network for deciphering cell-cell communication from spatial transcriptomics data.",
"tags": [
"preprocessing-tools"
],
"tasks": [
"Preprocessing"
],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "doubletfinder",
"name": "DoubletFinder",
"type": "toolkit",
"url": "https://github.com/chris-mcginnis-ucsf/DoubletFinder",
"description": "Machine learning approach for detecting multiplet (doublet) artifacts in single-cell RNA-seq data.",
"tags": [
"preprocessing-tools"
],
"tasks": [
"Preprocessing"
],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "flashdeconv",
"name": "FlashDeconv",
"type": "toolkit",
"url": "https://github.com/cafferychen777/flashdeconv",
"description": "High-performance spatial transcriptomics deconvolution (~1M spots in ~3 min).",
"tags": [
"preprocessing-tools"
],
"tasks": [
"Preprocessing"
],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "gromacs",
"name": "GROMACS",
"type": "toolkit",
"url": "https://www.gromacs.org/",
"description": "Molecular dynamics simulation package for biochemical molecules.",
"tags": [
"preprocessing-tools"
],
"tasks": [
"Preprocessing"
],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "harmony",
"name": "Harmony",
"type": "toolkit",
"url": "https://github.com/immunogenomics/harmony",
"description": "Fast and scalable integration of single-cell data across datasets, conditions, technologies, and species.",
"tags": [
"preprocessing-tools"
],
"tasks": [
"Preprocessing"
],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "kallisto",
"name": "kallisto",
"type": "toolkit",
"url": "https://pachterlab.github.io/kallisto/",
"description": "Near-optimal RNA-seq quantification using pseudoalignment for fast transcript abundance estimation.",
"tags": [
"preprocessing-tools"
],
"tasks": [
"Preprocessing"
],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "linger",
"name": "LINGER",
"type": "toolkit",
"url": "https://github.com/Durenlab/LINGER",
"description": "Neural network for gene regulatory network inference from single-cell multiome (RNA+ATAC-seq) data with bulk data pretraining.",
"tags": [
"preprocessing-tools"
],
"tasks": [
"Preprocessing"
],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "mdanalysis",
"name": "MDAnalysis",
"type": "toolkit",
"url": "https://www.mdanalysis.org/",
"description": "Python library for analyzing and altering molecular dynamics simulation trajectories.",
"tags": [
"preprocessing-tools"
],
"tasks": [
"Preprocessing"
],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "mogonet",
"name": "MOGONET",
"type": "toolkit",
"url": "https://github.com/txWang/MOGONET",
"description": "Multi-omics graph convolutional network framework for patient classification and biomarker identification.",
"tags": [
"preprocessing-tools"
],
"tasks": [
"Preprocessing"
],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "monocle3",
"name": "Monocle3",
"type": "toolkit",
"url": "https://cole-trapnell-lab.github.io/monocle3/",
"description": "Single-cell trajectory analysis tool for learning developmental trajectories and ordering cells in pseudotime.",
"tags": [
"preprocessing-tools"
],
"tasks": [
"Preprocessing"
],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "ncem",
"name": "NCEM",
"type": "toolkit",
"url": "https://github.com/theislab/ncem",
"description": "GNN-based model for learning intercellular communication from spatial graphs of cells.",
"tags": [
"preprocessing-tools"
],
"tasks": [
"Preprocessing"
],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "numbat",
"name": "Numbat",
"type": "toolkit",
"url": "https://github.com/kharchenkolab/numbat",
"description": "Haplotype-aware copy number variation inference from single-cell RNA-seq using hidden Markov models.",
"tags": [
"preprocessing-tools"
],
"tasks": [
"Preprocessing"
],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "openmm",
"name": "OpenMM",
"type": "toolkit",
"url": "https://openmm.org/",
"description": "High-performance toolkit for molecular simulation and GPU-accelerated MD.",
"tags": [
"preprocessing-tools"
],
"tasks": [
"Preprocessing"
],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "rdkit",
"name": "RDKit",
"type": "toolkit",
"url": "https://github.com/rdkit/rdkit",
"description": "Cheminformatics software & machine learning toolkit.",
"tags": [
"preprocessing-tools"
],
"tasks": [
"Preprocessing"
],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "scanpy",
"name": "Scanpy",
"type": "toolkit",
"url": "https://scanpy.readthedocs.io/en/stable/",
"description": "Python library for scRNA-seq analysis.",
"tags": [
"preprocessing-tools"
],
"tasks": [
"Preprocessing"
],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "scenic",
"name": "SCENIC",
"type": "toolkit",
"url": "https://github.com/aertslab/SCENIC",
"description": "Single-cell regulatory network inference and clustering linking transcription factors to co-expressed gene modules.",
"tags": [
"preprocessing-tools"
],
"tasks": [
"Preprocessing"
],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "scipenn",
"name": "sciPENN",
"type": "toolkit",
"url": "https://github.com/jlakkis/sciPENN",
"description": "RNN-based method for simultaneous protein expression prediction, uncertainty estimation, and cell-type label transfer from CITE-seq and scRNA-seq data.",
"tags": [
"preprocessing-tools"
],
"tasks": [
"Preprocessing"
],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "scvelo",
"name": "scVelo",
"type": "toolkit",
"url": "https://github.com/theislab/scvelo",
"description": "RNA velocity estimation for single-cell transcriptomics, inferring the direction and speed of cell differentiation.",
"tags": [
"preprocessing-tools"
],
"tasks": [
"Preprocessing"
],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "scvi_tools",
"name": "scvi-tools",
"type": "toolkit",
"url": "https://scvi-tools.org/",
"description": "Probabilistic models for single-cell omics data analysis.",
"tags": [
"preprocessing-tools"
],
"tasks": [
"Preprocessing"
],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "seqbench",
"name": "SeqBench",
"type": "toolkit",
"url": "https://seqbench.com/",
"description": "Web-based molecular biology sequence workbench for primer design, cloning simulation (Gibson, Golden Gate, restriction digest), CRISPR guide RNA design, and sequence analysis, with a public REST API, OpenAPI 3.1 spec, and MCP server.",
"tags": [
"preprocessing-tools"
],
"tasks": [
"Preprocessing"
],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "seurat",
"name": "Seurat",
"type": "toolkit",
"url": "https://satijalab.org/seurat/",
"description": "R library for scRNA-seq analysis.",
"tags": [
"preprocessing-tools"
],
"tasks": [
"Preprocessing"
],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "squidpy",
"name": "Squidpy",
"type": "toolkit",
"url": "https://squidpy.readthedocs.io/",
"description": "Python library for spatial single-cell analysis.",
"tags": [
"preprocessing-tools"
],
"tasks": [
"Preprocessing"
],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "stagate",
"name": "STAGATE",
"type": "toolkit",
"url": "https://github.com/RucDongLab/STAGATE",
"description": "Adaptive graph attention auto-encoder for spatial domain identification in spatial transcriptomics.",
"tags": [
"preprocessing-tools"
],
"tasks": [
"Preprocessing"
],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "star",
"name": "STAR",
"type": "toolkit",
"url": "https://github.com/alexdobin/STAR",
"description": "Ultrafast universal RNA-seq aligner with support for spliced alignment and single-cell quantification via STARsolo.",
"tags": [
"preprocessing-tools"
],
"tasks": [
"Preprocessing"
],
"modalities": [],
"organism": [],
"api": false
},
{
"id": "tigon",
"name": "TIGON",
"type": "toolkit",
"url": "https://github.com/yutongo/TIGON",
"description": "Neural optimal transport method for reconstructing growth and dynamic trajectories from single-cell transcriptomics.",
"tags": [
"preprocessing-tools"
],
"tasks": [
"Preprocessing"
],
"modalities": [],
"organism": [],
"api": false
}
]