--- title: "Resources" task: "" lineage_type: import upstream_source: https://github.com/inoue0426/awesome-computational-biology/blob/c6f07d90/data/resources.json upstream_sha: c6f07d90 imported_at: 2026-08-31 prompt_class: catalogue upstream_changes: accepted author: upstream validated: false --- [ { "id": "chembl_web_services", "name": "ChEMBL Web Services", "type": "api", "url": "https://www.ebi.ac.uk/chembl/ws", "description": "REST API for bioactive molecules, targets, and bioassays.", "tags": [ "api" ], "tasks": [], "modalities": [ "chemical-structure" ], "organism": [], "api": true, "entities": [ "molecule", "protein" ], "documentation": "https://www.ebi.ac.uk/chembl/api/data/docs", "last_checked": "2026-08-08", "metadata_sources": [ "https://www.ebi.ac.uk/chembl/api/data/docs" ] }, { "id": "clinicaltrials_gov_api", "name": "ClinicalTrials.gov API", "type": "api", "url": "https://clinicaltrials.gov/api/gui", "description": "API for querying clinical trial metadata and results.", "tags": [ "api" ], "tasks": [], "modalities": [ "clinical" ], "organism": [], "api": true, "entities": [ "disease", "drug" ], "documentation": "https://clinicaltrials.gov/data-api/api", "last_checked": "2026-08-08", "metadata_sources": [ "https://clinicaltrials.gov/data-api/api" ] }, { "id": "ensembl_rest_api", "name": "Ensembl REST API", "type": "api", "url": "https://rest.ensembl.org/", "description": "API for genomic annotations, variants, genes, and comparative genomics.", "tags": [ "api" ], "tasks": [], "modalities": [ "genomics" ], "organism": [], "api": true, "entities": [ "gene", "genome", "transcript", "variant" ], "documentation": "https://rest.ensembl.org/", "last_checked": "2026-08-08", "metadata_sources": [ "https://rest.ensembl.org/" ] }, { "id": "kegg_rest_api", "name": "KEGG REST API", "type": "api", "url": "https://www.kegg.jp/kegg/rest/keggapi.html", "description": "API for accessing KEGG pathways, compounds, genes, and reactions.", "tags": [ "api" ], "tasks": [], "modalities": [], "organism": [], "api": true, "entities": [ "compound", "gene", "pathway" ], "documentation": "https://www.kegg.jp/kegg/rest/keggapi.html", "last_checked": "2026-08-08", "metadata_sources": [ "https://www.kegg.jp/kegg/rest/keggapi.html" ] }, { "id": "ncbi_e_utilities", "name": "NCBI E-utilities", "type": "api", "url": "https://www.ncbi.nlm.nih.gov/books/NBK25501/", "description": "Unified APIs for accessing NCBI databases (Gene, GEO, SRA, PubChem, etc).", "tags": [ "api" ], "tasks": [], "modalities": [ "genomics", "transcriptomics" ], "organism": [], "api": true, "entities": [ "gene", "genome", "protein", "transcript", "variant" ], "documentation": "https://www.ncbi.nlm.nih.gov/books/NBK25501/", "last_checked": "2026-08-08", "metadata_sources": [ "https://www.ncbi.nlm.nih.gov/books/NBK25501/" ] }, { "id": "open_targets_platform_api", "name": "Open Targets Platform API", "type": "api", "url": "https://platform.opentargets.org/api", "description": "API for target–disease associations integrating genetics, genomics, and drug data.", "tags": [ "api" ], "tasks": [], "modalities": [ "genomics", "knowledge-graph" ], "organism": [], "api": true, "entities": [ "disease", "drug", "gene", "variant" ], "documentation": "https://platform.opentargets.org/api", "last_checked": "2026-08-08", "metadata_sources": [ "https://platform.opentargets.org/api" ] }, { "id": "pubmed_e_utilities_esearch_efetch", "name": "PubMed E-utilities (esearch/efetch)", "type": "api", "url": "https://www.nlm.nih.gov/dataguide/edirect/esearch.html", "description": "APIs for searching and retrieving biomedical literature from PubMed.", "tags": [ "api" ], "tasks": [], "modalities": [], "organism": [], "api": true, "documentation": "https://www.ncbi.nlm.nih.gov/books/NBK25501/", "last_checked": "2026-08-08", "metadata_sources": [ "https://www.ncbi.nlm.nih.gov/books/NBK25501/" ] }, { "id": "uniprot_rest_api", "name": "UniProt REST API", "type": "api", "url": "https://www.uniprot.org/help/api", "description": "Programmatic access to protein sequence and functional annotation data.", "tags": [ "api" ], "tasks": [], "modalities": [ "protein-sequence", "proteomics" ], "organism": [], "api": true, "entities": [ "protein" ], "documentation": "https://www.uniprot.org/help/api", "last_checked": "2026-08-08", "metadata_sources": [ "https://www.uniprot.org/help/api" ] }, { "id": "1000_genomes_project", "name": "1000 Genomes Project", "type": "benchmark", "url": "https://www.internationalgenome.org/", "description": "Reference panel of human genetic variation from 2,504 individuals across 26 populations.", "tags": [ "benchmarks-and-datasets" ], "tasks": [], "modalities": [], "organism": [], "api": false }, { "id": "bace", "name": "BACE", "type": "benchmark", "url": "https://www.kaggle.com/datasets/gokturkkoch/bace", "description": "Binary classification and regression dataset for β-secretase 1 (BACE-1) inhibitor binding affinity.", "tags": [ "benchmarks-and-datasets" ], "tasks": [ "classification", "regression" ], "modalities": [ "chemical-structure" ], "organism": [], "api": false, "entities": [ "molecule", "protein" ], "last_checked": "2026-08-08", "metadata_sources": [ "https://www.kaggle.com/datasets/gokturkkoch/bace" ] }, { "id": "beat_aml", "name": "BEAT AML", "type": "benchmark", "url": "https://biodev.github.io/BeatAML2/", "description": "Functional ex vivo drug sensitivity measurements paired with genomics for acute myeloid leukemia.", "tags": [ "benchmarks-and-datasets" ], "tasks": [ "drug-response-prediction" ], "modalities": [ "genomics" ], "organism": [], "api": false, "entities": [ "cell", "disease", "drug", "gene" ], "last_checked": "2026-08-08", "metadata_sources": [ "https://biodev.github.io/BeatAML2/" ] }, { "id": "bento", "name": "Bento", "type": "benchmark", "url": "https://github.com/LigandPro/Bento", "description": "Protein-ligand docking benchmark covering rigid, flexible, de novo, blind, induced-fit, and covalent docking tasks.", "tags": [ "benchmarks-and-datasets" ], "tasks": [], "modalities": [], "organism": [], "api": false }, { "id": "bindingdb_curated_sets", "name": "BindingDB Curated Sets", "type": "benchmark", "url": "https://www.bindingdb.org/rwd/bind/chemsearch/marvin/SDFdownload.jsp?all_download=yes", "description": "Curated binding affinity datasets for protein–ligand interaction benchmarking.", "tags": [ "benchmarks-and-datasets" ], "tasks": [ "drug-target-interaction" ], "modalities": [ "chemical-structure" ], "organism": [], "api": false, "entities": [ "molecule", "protein" ], "last_checked": "2026-08-08", "metadata_sources": [ "https://www.bindingdb.org/" ] }, { "id": "cancer_therapeutics_response_portal_ctrp", "name": "Cancer Therapeutics Response Portal (CTRP)", "type": "benchmark", "url": "https://portals.broadinstitute.org/ctrp/", "description": "Drug sensitivity profiles across ~900 cancer cell lines for >400 compounds.", "tags": [ "benchmarks-and-datasets" ], "tasks": [ "drug-response-prediction" ], "modalities": [], "organism": [], "api": false, "entities": [ "cell", "drug" ], "last_checked": "2026-08-08", "metadata_sources": [ "https://portals.broadinstitute.org/ctrp/" ] }, { "id": "clintox", "name": "ClinTox", "type": "benchmark", "url": "https://tdcommons.ai/single_pred_tasks/tox/#clintox", "description": "Clinical toxicity dataset contrasting FDA-approved drugs with those that failed clinical trials due to toxicity.", "tags": [ "benchmarks-and-datasets" ], "tasks": [ "classification" ], "modalities": [ "clinical" ], "organism": [], "api": false, "entities": [ "drug" ], "last_checked": "2026-08-08", "metadata_sources": [ "https://tdcommons.ai/single_pred_tasks/tox/#clintox" ] }, { "id": "cptac_clinical_proteomic_tumor_analysis_consortium", "name": "CPTAC (Clinical Proteomic Tumor Analysis Consortium)", "type": "benchmark", "url": "https://proteomics.cancer.gov/programs/cptac", "description": "Multi-omic proteogenomic datasets for multiple cancer types linking proteomics with genomics.", "tags": [ "benchmarks-and-datasets" ], "tasks": [], "modalities": [], "organism": [], "api": false }, { "id": "crossdocked2020", "name": "CrossDocked2020", "type": "benchmark", "url": "https://arxiv.org/abs/2001.01037", "description": "Large-scale dataset for structure-based virtual screening.", "tags": [ "benchmarks-and-datasets" ], "tasks": [], "modalities": [], "organism": [], "api": false }, { "id": "dud_e_directory_of_useful_decoys_enhanced", "name": "DUD-E (Directory of Useful Decoys, Enhanced)", "type": "benchmark", "url": "http://dude.docking.org/", "description": "Structure-based virtual screening benchmark with active ligands and challenging decoy sets across diverse protein targets.", "tags": [ "benchmarks-and-datasets" ], "tasks": [], "modalities": [], "organism": [], "api": false }, { "id": "flip_fitness_landscape_inference_for_proteins", "name": "FLIP (Fitness Landscape Inference for Proteins)", "type": "benchmark", "url": "https://github.com/J-SNACKKB/FLIP", "description": "Benchmark collection of protein fitness landscape datasets for evaluating protein ML models.", "tags": [ "benchmarks-and-datasets" ], "tasks": [], "modalities": [], "organism": [], "api": false }, { "id": "guacamol", "name": "GuacaMol", "type": "benchmark", "url": "https://github.com/BenevolentAI/guacamol", "description": "Benchmark suite for generative molecular design models.", "tags": [ "benchmarks-and-datasets" ], "tasks": [ "molecular-generation" ], "modalities": [ "chemical-structure" ], "organism": [], "api": false, "entities": [ "molecule" ], "github": "https://github.com/BenevolentAI/guacamol", "last_checked": "2026-08-08", "metadata_sources": [ "https://github.com/BenevolentAI/guacamol" ] }, { "id": "hest_xenium_virtual_spatial_transcriptomics", "name": "HEST Xenium virtual spatial transcriptomics", "type": "benchmark", "url": "https://huggingface.co/datasets/ratschlab/HEST_Xenium_virtual_spatial_transcriptomics", "description": "DeepSpot-M predicted transcriptome-wide ST for 59 HEST-1k 10x Xenium samples (~13.3M cells) (gated). Paper: [DeepSpot-M](https://www.medrxiv.org/content/10.64898/2026.06.19.26356060v1).", "tags": [ "benchmarks-and-datasets" ], "tasks": [ "regression" ], "modalities": [ "histopathology", "spatial-transcriptomics", "transcriptomics" ], "organism": [], "api": false, "entities": [ "cell", "gene", "tissue" ], "documentation": "https://huggingface.co/datasets/ratschlab/HEST_Xenium_virtual_spatial_transcriptomics", "last_checked": "2026-08-08", "metadata_sources": [ "https://huggingface.co/datasets/ratschlab/HEST_Xenium_virtual_spatial_transcriptomics" ] }, { "id": "jump_cell_painting_datasets", "name": "JUMP Cell Painting Datasets", "type": "benchmark", "url": "https://github.com/jump-cellpainting/datasets", "description": "Consortium-scale cell imaging perturbation datasets (chemical and genetic) for phenotypic profiling and drug discovery research.", "tags": [ "benchmarks-and-datasets" ], "tasks": [], "modalities": [], "organism": [], "api": false }, { "id": "lincs_l1000", "name": "LINCS L1000", "type": "benchmark", "url": "https://lincsproject.org/LINCS/tools/workflows/find-the-best-place-to-obtain-the-lincs-l1000-data", "description": "Gene expression profiles (978 landmark genes) for >20,000 chemical and genetic perturbations across cell lines.", "tags": [ "benchmarks-and-datasets" ], "tasks": [ "perturbation-prediction" ], "modalities": [ "transcriptomics" ], "organism": [], "api": false, "entities": [ "cell", "compound", "gene" ], "last_checked": "2026-08-08", "metadata_sources": [ "https://lincsproject.org/LINCS/tools/workflows/find-the-best-place-to-obtain-the-lincs-l1000-data" ] }, { "id": "moleculenet", "name": "MoleculeNet", "type": "benchmark", "url": "http://moleculenet.ai/", "description": "Benchmark datasets for molecular machine learning.", "tags": [ "benchmarks-and-datasets" ], "tasks": [ "classification", "regression" ], "modalities": [ "chemical-structure" ], "organism": [], "api": false, "entities": [ "molecule" ], "github": "https://github.com/deepchem/moleculenet", "last_checked": "2026-08-08", "metadata_sources": [ "https://github.com/deepchem/moleculenet" ] }, { "id": "moses", "name": "MOSES", "type": "benchmark", "url": "https://github.com/molecularsets/moses", "description": "Benchmarking platform for molecular generation models.", "tags": [ "benchmarks-and-datasets" ], "tasks": [], "modalities": [], "organism": [], "api": false }, { "id": "ogb_open_graph_benchmark", "name": "OGB (Open Graph Benchmark)", "type": "benchmark", "url": "https://ogb.stanford.edu/", "description": "Large-scale graph ML benchmark suite including biological datasets such as ogbl-ppa (protein-protein associations) and ogbg-molhiv.", "tags": [ "benchmarks-and-datasets" ], "tasks": [], "modalities": [], "organism": [], "api": false }, { "id": "openbiolink", "name": "OpenBioLink", "type": "benchmark", "url": "https://github.com/OpenBioLink/OpenBioLink", "description": "Benchmark datasets for biological knowledge graph completion.", "tags": [ "benchmarks-and-datasets" ], "tasks": [], "modalities": [], "organism": [], "api": false }, { "id": "pharmgkb", "name": "PharmGKB", "type": "benchmark", "url": "https://www.pharmgkb.org/", "description": "Curated pharmacogenomics dataset linking genetic variants to drug response phenotypes across thousands of drugs.", "tags": [ "benchmarks-and-datasets" ], "tasks": [ "drug-response-prediction" ], "modalities": [ "clinical", "genomics" ], "organism": [], "api": false, "entities": [ "drug", "gene", "phenotype", "variant" ], "last_checked": "2026-08-08", "metadata_sources": [ "https://www.pharmgkb.org/" ] }, { "id": "pk_db", "name": "PK-DB", "type": "benchmark", "url": "https://pk-db.com/", "description": "Open database of experimental pharmacokinetics (PK) and ADME data from clinical and preclinical studies.", "tags": [ "benchmarks-and-datasets" ], "tasks": [], "modalities": [ "clinical" ], "organism": [], "api": false, "entities": [ "drug" ], "last_checked": "2026-08-08", "metadata_sources": [ "https://pk-db.com/" ] }, { "id": "prism", "name": "PRISM", "type": "benchmark", "url": "https://depmap.org/portal/prism/", "description": "Cancer drug sensitivity profiling of >4,500 drugs across >900 cancer cell lines using pooled-cell-line barcoding.", "tags": [ "benchmarks-and-datasets" ], "tasks": [ "drug-response-prediction" ], "modalities": [], "organism": [], "api": false, "entities": [ "cell", "drug" ], "last_checked": "2026-08-08", "metadata_sources": [ "https://depmap.org/portal/prism/" ] }, { "id": "proteingym", "name": "ProteinGym", "type": "benchmark", "url": "https://github.com/OATML-Markslab/ProteinGym", "description": "Large-scale benchmark of deep mutational scanning assays for evaluating protein fitness landscape models.", "tags": [ "benchmarks-and-datasets" ], "tasks": [ "regression" ], "modalities": [ "protein-sequence" ], "organism": [], "api": false, "entities": [ "protein" ], "github": "https://github.com/OATML-Markslab/ProteinGym", "last_checked": "2026-08-08", "metadata_sources": [ "https://github.com/OATML-Markslab/ProteinGym" ] }, { "id": "qm9", "name": "QM9", "type": "benchmark", "url": "https://figshare.com/collections/Quantum_chemistry_structures_and_properties_of_134_kilo_molecules/978904", "description": "Quantum chemistry properties for 134K stable small organic molecules computed at DFT level.", "tags": [ "benchmarks-and-datasets" ], "tasks": [], "modalities": [], "organism": [], "api": false }, { "id": "scib_single_cell_integration_benchmarks", "name": "scIB (Single-cell Integration Benchmarks)", "type": "benchmark", "url": "https://github.com/theislab/scib", "description": "Comprehensive benchmarking framework for single-cell data integration methods.", "tags": [ "benchmarks-and-datasets" ], "tasks": [], "modalities": [], "organism": [], "api": false }, { "id": "scperturb", "name": "scPerturb", "type": "benchmark", "url": "https://github.com/sanderlab/scPerturb", "description": "Curated and continuously updated single-cell perturbation data resource spanning CRISPR and drug perturbation studies.", "tags": [ "benchmarks-and-datasets" ], "tasks": [ "perturbation-prediction" ], "modalities": [ "single-cell-rna-seq" ], "organism": [], "api": false, "entities": [ "cell", "drug", "gene" ], "github": "https://github.com/sanderlab/scPerturb", "last_checked": "2026-08-08", "metadata_sources": [ "https://github.com/sanderlab/scPerturb" ] }, { "id": "sider_side_effect_resource", "name": "SIDER (Side Effect Resource)", "type": "benchmark", "url": "http://sideeffects.embl.de/", "description": "Database of 1,430 approved drugs with their recorded adverse drug reactions across 27 system-organ classes.", "tags": [ "benchmarks-and-datasets" ], "tasks": [], "modalities": [ "clinical" ], "organism": [], "api": false, "entities": [ "drug", "phenotype" ], "last_checked": "2026-08-08", "metadata_sources": [ "http://sideeffects.embl.de/" ] }, { "id": "tabula_muris", "name": "Tabula Muris", "type": "benchmark", "url": "https://tabula-muris.ds.czbiohub.org/", "description": "Comprehensive single-cell atlas of 20 mouse organs and tissues, enabling cross-tissue and cross-species comparisons.", "tags": [ "benchmarks-and-datasets" ], "tasks": [], "modalities": [], "organism": [], "api": false }, { "id": "tabula_sapiens", "name": "Tabula Sapiens", "type": "benchmark", "url": "https://tabula-sapiens-portal.ds.czbiohub.org/", "description": "Comprehensive human single-cell atlas of ~500K cells from 24 organs and tissues across multiple donors.", "tags": [ "benchmarks-and-datasets" ], "tasks": [], "modalities": [], "organism": [], "api": false }, { "id": "tape_tasks_assessing_protein_embeddings", "name": "TAPE (Tasks Assessing Protein Embeddings)", "type": "benchmark", "url": "https://github.com/songlab-cal/tape", "description": "Benchmark suite of five biologically meaningful semi-supervised learning tasks for evaluating protein representations.", "tags": [ "benchmarks-and-datasets" ], "tasks": [], "modalities": [], "organism": [], "api": false }, { "id": "tcga_virtual_spatial_transcriptomics_atlas", "name": "TCGA virtual spatial transcriptomics atlas", "type": "benchmark", "url": "https://huggingface.co/datasets/ratschlab/TCGA_virtual_spatial_transcriptomics_atlas", "description": "DeepSpot-M predicted transcriptome-wide ST for TCGA H&E (FF + FFPE; 28,664 slides / 32 cancer types; gated). Paper: [DeepSpot-M](https://www.medrxiv.org/content/10.64898/2026.06.19.26356060v1).", "tags": [ "benchmarks-and-datasets" ], "tasks": [], "modalities": [], "organism": [], "api": false }, { "id": "the_cancer_genome_atlas_tcga", "name": "The Cancer Genome Atlas (TCGA)", "type": "benchmark", "url": "https://www.cancer.gov/about-nci/organization/ccg/research/structural-genomics/tcga", "description": "Comprehensive multi-omics (genomics, transcriptomics, proteomics, methylation) dataset for 33 cancer types across ~11,000 patients.", "tags": [ "benchmarks-and-datasets" ], "tasks": [], "modalities": [], "organism": [], "api": false }, { "id": "therapeutics_data_commons_tdc", "name": "Therapeutics Data Commons (TDC)", "type": "benchmark", "url": "https://tdcommons.ai/", "description": "Unified benchmark suite covering ADMET, drug-target interaction, drug response, and more.", "tags": [ "benchmarks-and-datasets" ], "tasks": [], "modalities": [], "organism": [], "api": false }, { "id": "tox21", "name": "Tox21", "type": "benchmark", "url": "https://tripod.nih.gov/tox21/challenge/", "description": "12,707 compounds tested in 12 nuclear receptor and stress-response pathway biochemical assays for toxicity prediction.", "tags": [ "benchmarks-and-datasets" ], "tasks": [], "modalities": [], "organism": [], "api": false }, { "id": "uk_biobank", "name": "UK Biobank", "type": "benchmark", "url": "https://www.ukbiobank.ac.uk/", "description": "Large-scale biomedical database of ~500K participants with genetic, imaging, and health data for population genetics and disease studies.", "tags": [ "benchmarks-and-datasets" ], "tasks": [], "modalities": [], "organism": [], "api": false }, { "id": "10x_genomics_dataset", "name": "10x Genomics Dataset", "type": "database", "url": "https://www.10xgenomics.com/resources/datasets", "description": "Collection of single-cell datasets.", "tags": [ "genome" ], "tasks": [], "modalities": [ "single-cell-rna-seq", "genomics" ], "organism": [], "api": false, "entities": [ "cell", "gene" ], "documentation": "https://www.10xgenomics.com/resources/datasets", "last_checked": "2026-08-08", "metadata_sources": [ "https://www.10xgenomics.com/resources/datasets" ] }, { "id": "alphafold_protein_structure_database", "name": "AlphaFold Protein Structure Database", "type": "database", "url": "https://alphafold.ebi.ac.uk/api-docs", "description": "3D protein structure predictions.", "tags": [ "protein" ], "tasks": [], "modalities": [ "molecular-structure", "protein-sequence" ], "organism": [], "api": false, "entities": [ "protein" ], "documentation": "https://alphafold.ebi.ac.uk/api-docs", "last_checked": "2026-08-08", "metadata_sources": [ "https://alphafold.ebi.ac.uk/", "https://alphafold.ebi.ac.uk/api-docs" ] }, { "id": "bindingdb", "name": "BindingDB", "type": "database", "url": "https://www.bindingdb.org/rwd/bind/index.jsp", "description": "Compounds and target database.", "tags": [ "chemical-protein-interaction", "interaction" ], "tasks": [], "modalities": [ "chemical-structure", "molecular-structure" ], "organism": [], "api": false, "entities": [ "molecule", "protein" ], "documentation": "https://www.bindingdb.org/rwd/bind/index.jsp", "last_checked": "2026-08-08", "metadata_sources": [ "https://www.bindingdb.org/rwd/bind/index.jsp" ] }, { "id": "biocyc", "name": "BioCyc", "type": "database", "url": "https://biocyc.org/", "description": "Collection of pathway/genome databases across thousands of organisms.", "tags": [ "pathway" ], "tasks": [], "modalities": [ "genomics" ], "organism": [], "api": false, "entities": [ "gene", "pathway", "organism" ], "documentation": "https://biocyc.org/", "last_checked": "2026-08-08", "metadata_sources": [ "https://biocyc.org/" ] }, { "id": "biogrid", "name": "BioGRID", "type": "database", "url": "https://thebiogrid.org/", "description": "Protein, genetic, and chemical interactions.", "tags": [ "interaction", "protein-protein-interaction" ], "tasks": [], "modalities": [ "Protein" ], "organism": [], "api": false, "entities": [ "gene", "protein" ], "documentation": "https://thebiogrid.org/", "last_checked": "2026-08-08", "metadata_sources": [ "https://thebiogrid.org/" ] }, { "id": "cancer_cell_line_encyclopedia", "name": "Cancer Cell Line Encyclopedia", "type": "database", "url": "https://sites.broadinstitute.org/ccle/", "description": "Database of ~1000 cancer cell lines.", "tags": [ "drug-cell-line-response", "interaction" ], "tasks": [], "modalities": [ "genomics", "transcriptomics" ], "organism": [], "api": false, "entities": [ "cell", "gene", "disease", "drug" ], "documentation": "https://sites.broadinstitute.org/ccle/", "last_checked": "2026-08-08", "metadata_sources": [ "https://sites.broadinstitute.org/ccle/" ] }, { "id": "catalogue_of_somatic_mutations_in_cancer_cosmic", "name": "Catalogue Of Somatic Mutations In Cancer (COSMIC)", "type": "database", "url": "https://cancer.sanger.ac.uk/cosmic", "description": "Resource on somatic mutations in cancers.", "tags": [ "genome" ], "tasks": [], "modalities": [ "genomics" ], "organism": [], "api": false, "entities": [ "disease", "gene", "variant" ], "documentation": "https://cancer.sanger.ac.uk/cosmic", "last_checked": "2026-08-08", "metadata_sources": [ "https://cancer.sanger.ac.uk/cosmic" ] }, { "id": "cath_database", "name": "CATH database", "type": "database", "url": "https://www.cathdb.info/", "description": "Hierarchical classification of protein domain structures.", "tags": [ "protein" ], "tasks": [], "modalities": [ "molecular-structure", "protein-sequence" ], "organism": [], "api": false, "entities": [ "protein" ], "documentation": "https://www.cathdb.info/", "last_checked": "2026-08-08", "metadata_sources": [ "https://www.cathdb.info/" ] }, { "id": "cbioportal", "name": "cBioPortal", "type": "database", "url": "https://www.cbioportal.org/", "description": "Cancer genomics database; aggregating many patient datasets.", "tags": [ "genome" ], "tasks": [], "modalities": [ "genomics", "clinical" ], "organism": [], "api": false, "entities": [ "disease", "gene", "variant" ], "github": "https://github.com/cBioPortal/cbioportal", "documentation": "https://www.cbioportal.org/", "last_checked": "2026-08-08", "metadata_sources": [ "https://www.cbioportal.org/", "https://github.com/cBioPortal/cbioportal" ] }, { "id": "cellminer_cross_database_cellminercdb", "name": "CellMiner Cross Database (CellMinerCDB)", "type": "database", "url": "https://discover.nci.nih.gov/cellminercdb/", "description": "Integrates multiple cancer cell line databases.", "tags": [ "drug-cell-line-response", "interaction" ], "tasks": [ "drug-response-prediction" ], "modalities": [ "genomics" ], "organism": [], "api": false, "entities": [ "cell", "drug", "gene" ], "last_checked": "2026-08-08", "metadata_sources": [ "https://discover.nci.nih.gov/cellminercdb/" ] }, { "id": "chebi", "name": "ChEBI", "type": "database", "url": "https://www.ebi.ac.uk/chebi/", "description": "Database focused on small chemical compounds.", "tags": [ "compound" ], "tasks": [], "modalities": [ "Small Molecule" ], "organism": [], "api": false }, { "id": "chembl", "name": "ChEMBL", "type": "database", "url": "https://www.ebi.ac.uk/chembl/", "description": "Bioactive molecules with drug-like properties.", "tags": [ "compound" ], "tasks": [], "modalities": [ "chemical-structure" ], "organism": [], "api": false, "entities": [ "compound", "molecule", "protein" ], "documentation": "https://www.ebi.ac.uk/chembl/", "last_checked": "2026-08-08", "metadata_sources": [ "https://www.ebi.ac.uk/chembl/" ] }, { "id": "chemspider", "name": "ChemSpider", "type": "database", "url": "http://www.chemspider.com/", "description": "Chemical structure database.", "tags": [ "compound" ], "tasks": [], "modalities": [ "Small Molecule" ], "organism": [], "api": false }, { "id": "clinicaltrials_gov", "name": "ClinicalTrials.gov", "type": "database", "url": "https://clinicaltrials.gov/", "description": "Privately and publicly funded clinical studies.", "tags": [ "clinical-trial" ], "tasks": [], "modalities": [ "clinical" ], "organism": [], "api": false, "entities": [ "disease", "drug" ], "documentation": "https://clinicaltrials.gov/", "last_checked": "2026-08-08", "metadata_sources": [ "https://clinicaltrials.gov/" ] }, { "id": "comparative_toxicogenomics_database", "name": "Comparative Toxicogenomics Database", "type": "database", "url": "http://ctdbase.org/", "description": "Chemical-gene interactions, chemical-disease and gene-disease associations, chemical-phenotype associations.", "tags": [ "drug-gene-interaction", "interaction" ], "tasks": [], "modalities": [ "knowledge-graph" ], "organism": [], "api": false, "entities": [ "compound", "gene", "disease" ], "documentation": "https://ctdbase.org/", "last_checked": "2026-08-08", "metadata_sources": [ "https://ctdbase.org/" ] }, { "id": "critical_assessment_of_structure_prediction_casp", "name": "Critical Assessment of Structure Prediction (CASP)", "type": "database", "url": "https://predictioncenter.org/", "description": "Assessing methods for protein structure prediction.", "tags": [ "protein" ], "tasks": [], "modalities": [ "Protein" ], "organism": [], "api": false }, { "id": "cz_cellxgene", "name": "CZ CELLxGENE", "type": "database", "url": "https://cellxgene.cziscience.com/", "description": "Single-cell dataset repository and interactive explorer from the Chan Zuckerberg Initiative.", "tags": [ "scrna" ], "tasks": [], "modalities": [ "single-cell-rna-seq", "transcriptomics" ], "organism": [], "api": false, "entities": [ "cell", "gene", "tissue" ], "documentation": "https://cellxgene.cziscience.com/", "last_checked": "2026-08-08", "metadata_sources": [ "https://cellxgene.cziscience.com/" ] }, { "id": "davis_kinase_inhibitors_db", "name": "Davis kinase inhibitors DB", "type": "database", "url": "http://staff.cs.utu.fi/~aijrinas/dti/", "description": "Experimental kinase inhibitor binding affinity dataset for protein–ligand interaction research.", "tags": [ "chemical-protein-interaction", "interaction" ], "tasks": [], "modalities": [ "Protein", "Small Molecule" ], "organism": [], "api": false }, { "id": "dependency_map_depmap", "name": "Dependency Map (DepMap)", "type": "database", "url": "https://depmap.org/portal/", "description": "CRISPR-Cas9 screens in cancer cell lines.", "tags": [ "genome" ], "tasks": [], "modalities": [ "genomics" ], "organism": [], "api": false, "entities": [ "cell", "gene", "disease", "drug" ], "documentation": "https://depmap.org/portal/", "last_checked": "2026-08-08", "metadata_sources": [ "https://depmap.org/portal/" ] }, { "id": "dgidb", "name": "DGIdb", "type": "database", "url": "https://www.dgidb.org/", "description": "Drug-gene interactions and the druggable genome.", "tags": [ "drug-gene-interaction", "interaction" ], "tasks": [], "modalities": [ "knowledge-graph" ], "organism": [], "api": false, "entities": [ "drug", "gene" ], "documentation": "https://www.dgidb.org/", "last_checked": "2026-08-08", "metadata_sources": [ "https://www.dgidb.org/" ] }, { "id": "diseases", "name": "DISEASES", "type": "database", "url": "https://diseases.jensenlab.org/", "description": "Gene–disease association database integrating evidence from text mining, curated databases, and experimental data.", "tags": [ "disease" ], "tasks": [], "modalities": [ "Disease" ], "organism": [], "api": false }, { "id": "disgenet", "name": "DisGeNET", "type": "database", "url": "https://www.disgenet.org/", "description": "Database of gene-disease associations integrating expert-curated and GWAS data.", "tags": [ "disease" ], "tasks": [], "modalities": [ "Disease" ], "organism": [], "api": false }, { "id": "drkg", "name": "DRKG", "type": "database", "url": "https://github.com/gnn4dr/DRKG", "description": "Large-scale biological knowledge graph for drug discovery.", "tags": [ "interaction", "knowledge-graph" ], "tasks": [], "modalities": [ "Knowledge Graph" ], "organism": [], "api": false }, { "id": "drug_mechanism_database_drugmechdb", "name": "Drug Mechanism Database (DrugMechDB)", "type": "database", "url": "https://github.com/SuLab/DrugMechDB/tree/2.0.1", "description": "Mechanisms of action from drug to disease.", "tags": [ "interaction", "knowledge-graph" ], "tasks": [], "modalities": [ "knowledge-graph" ], "organism": [], "api": false, "entities": [ "drug", "disease", "gene", "pathway" ], "github": "https://github.com/SuLab/DrugMechDB", "last_checked": "2026-08-08", "metadata_sources": [ "https://github.com/SuLab/DrugMechDB" ] }, { "id": "drug_repurposing_hub", "name": "Drug Repurposing Hub", "type": "database", "url": "https://repo-hub.broadinstitute.org/repurposing#download-data", "description": "Collections of drug repurposing data (drug, MoA, target, etc).", "tags": [ "compound" ], "tasks": [], "modalities": [ "chemical-structure" ], "organism": [], "api": false, "entities": [ "drug", "protein", "disease" ], "documentation": "https://repo-hub.broadinstitute.org/repurposing", "last_checked": "2026-08-08", "metadata_sources": [ "https://repo-hub.broadinstitute.org/repurposing" ] }, { "id": "drugbank", "name": "DrugBank", "type": "database", "url": "https://go.drugbank.com/", "description": "Database of drugs and targets (University of Alberta).", "tags": [ "disease" ], "tasks": [], "modalities": [ "chemical-structure" ], "organism": [], "api": false, "entities": [ "disease", "drug", "protein" ], "last_checked": "2026-08-08", "metadata_sources": [ "https://go.drugbank.com/" ] }, { "id": "drugcentral", "name": "DrugCentral", "type": "database", "url": "http://drugcentral.org/", "description": "Online drug compendium with drug mode of action and indication information.", "tags": [ "compound" ], "tasks": [], "modalities": [ "chemical-structure" ], "organism": [], "api": false, "entities": [ "drug", "protein", "disease" ], "documentation": "https://drugcentral.org/", "last_checked": "2026-08-08", "metadata_sources": [ "https://drugcentral.org/" ] }, { "id": "drugtargetcommons", "name": "DrugTargetCommons", "type": "database", "url": "https://drugtargetcommons.fimm.fi/", "description": "Community platform for curating and integrating experimental bioactivity data across drugs and targets.", "tags": [ "compound" ], "tasks": [], "modalities": [ "chemical-structure" ], "organism": [], "api": false, "entities": [ "drug", "protein" ], "documentation": "https://drugtargetcommons.fimm.fi/", "last_checked": "2026-08-08", "metadata_sources": [ "https://drugtargetcommons.fimm.fi/" ] }, { "id": "encode", "name": "ENCODE", "type": "database", "url": "https://www.encodeproject.org/", "description": "Encyclopedia of DNA Elements; regulatory and functional genomic elements across the genome.", "tags": [ "genome" ], "tasks": [], "modalities": [ "Genomics" ], "organism": [], "api": false }, { "id": "ensembl", "name": "Ensembl", "type": "database", "url": "https://www.ensembl.org/", "description": "Genome browser and annotation database for vertebrate and other eukaryotic genomes.", "tags": [ "genome" ], "tasks": [], "modalities": [ "Genomics" ], "organism": [], "api": false }, { "id": "eu_drug_regulating_authorities_clinical_trials_db_eudract", "name": "EU Drug Regulating Authorities Clinical Trials DB (EudraCT)", "type": "database", "url": "https://eudract.ema.europa.eu/", "description": "European clinical trial database.", "tags": [ "clinical-trial" ], "tasks": [], "modalities": [ "Clinical" ], "organism": [], "api": false }, { "id": "fantom5", "name": "FANTOM5", "type": "database", "url": "https://fantom.gsc.riken.jp/5/", "description": "Functional annotation of mammalian genome; comprehensive atlas of active enhancers, promoters, and transcription start sites across human and mouse cell types.", "tags": [ "genome" ], "tasks": [], "modalities": [ "Genomics" ], "organism": [], "api": false }, { "id": "genbank", "name": "GenBank", "type": "database", "url": "https://www.ncbi.nlm.nih.gov/genbank/", "description": "NCBI's database of genetic sequences.", "tags": [ "genome" ], "tasks": [], "modalities": [ "Genomics" ], "organism": [], "api": false }, { "id": "gene_expression_omnibus", "name": "Gene Expression Omnibus", "type": "database", "url": "https://www.ncbi.nlm.nih.gov/geo/", "description": "Public functional genomics database.", "tags": [ "scrna" ], "tasks": [], "modalities": [ "Single Cell" ], "organism": [], "api": false }, { "id": "genomics_of_drug_sensitivity_in_cancer_gdsc", "name": "Genomics of Drug Sensitivity in Cancer (GDSC)", "type": "database", "url": "https://www.cancerrxgene.org/", "description": "Drug sensitivity for ~1000 human cancer cell lines and hundreds of compounds.", "tags": [ "benchmarks-and-datasets", "drug-cell-line-response", "interaction" ], "tasks": [ "drug-response-prediction" ], "modalities": [ "genomics" ], "organism": [], "api": false, "entities": [ "cell", "drug", "gene" ], "last_checked": "2026-08-08", "metadata_sources": [ "https://www.cancerrxgene.org/" ] }, { "id": "gnomad", "name": "gnomAD", "type": "database", "url": "https://gnomad.broadinstitute.org/", "description": "Genome Aggregation Database; genetic variation from large-scale sequencing projects.", "tags": [ "genome" ], "tasks": [], "modalities": [ "Genomics" ], "organism": [], "api": false }, { "id": "hetionet", "name": "Hetionet", "type": "database", "url": "https://github.com/hetio/hetionet", "description": "Heterogeneous network integrating genes, diseases, drugs, pathways, and more.", "tags": [ "interaction", "knowledge-graph" ], "tasks": [], "modalities": [ "Knowledge Graph" ], "organism": [], "api": false }, { "id": "hippie", "name": "HIPPIE", "type": "database", "url": "http://cbdm-01.zdv.uni-mainz.de/~mschaefer/hippie/", "description": "Human protein-protein interaction database.", "tags": [ "interaction", "protein-protein-interaction" ], "tasks": [], "modalities": [ "Protein" ], "organism": [], "api": false }, { "id": "hmdb_human_metabolome_database", "name": "HMDB (Human Metabolome Database)", "type": "database", "url": "https://hmdb.ca/", "description": "Comprehensive database of small molecule metabolites found in the human body.", "tags": [ "compound" ], "tasks": [], "modalities": [ "Small Molecule" ], "organism": [], "api": false }, { "id": "human_cell_atlas", "name": "Human Cell Atlas", "type": "database", "url": "https://www.humancellatlas.org/", "description": "Open global atlas of all cells in the human body.", "tags": [ "scrna" ], "tasks": [], "modalities": [ "Single Cell" ], "organism": [], "api": false }, { "id": "human_genome_resources_at_ncbi", "name": "Human Genome Resources at NCBI", "type": "database", "url": "https://www.ncbi.nlm.nih.gov/projects/genome/guide/human/index.shtml", "description": "Database for genomics, proteomics, transcriptomics, and systems biology.", "tags": [ "genome" ], "tasks": [], "modalities": [ "Genomics" ], "organism": [], "api": false }, { "id": "human_phenotype_ontology_hpo", "name": "Human Phenotype Ontology (HPO)", "type": "database", "url": "https://hpo.jax.org/", "description": "Standardized vocabulary of phenotypic abnormalities in human disease, linking genes, variants, and clinical features.", "tags": [ "disease" ], "tasks": [], "modalities": [ "Disease" ], "organism": [], "api": false }, { "id": "icd10", "name": "ICD10", "type": "database", "url": "https://icd.who.int/browse10/2019/en", "description": "International Classification of Diseases, 10th revision.", "tags": [ "clinical-trial" ], "tasks": [], "modalities": [ "Clinical" ], "organism": [], "api": false }, { "id": "intact", "name": "IntAct", "type": "database", "url": "https://www.ebi.ac.uk/intact/home", "description": "Open-source molecular interaction database and analysis system from EMBL-EBI.", "tags": [ "interaction", "protein-protein-interaction" ], "tasks": [], "modalities": [ "Protein" ], "organism": [], "api": false }, { "id": "interpro", "name": "InterPro", "type": "database", "url": "https://www.ebi.ac.uk/interpro/", "description": "Protein families, domains, and functional sites database integrating 14 member databases including Pfam and PROSITE.", "tags": [ "protein" ], "tasks": [], "modalities": [ "Protein" ], "organism": [], "api": false }, { "id": "jaspar", "name": "JASPAR", "type": "database", "url": "http://jaspar.genereg.net/", "description": "Database of transcription factor binding profiles.", "tags": [ "genome" ], "tasks": [], "modalities": [ "Genomics" ], "organism": [], "api": false }, { "id": "kegg_compound", "name": "KEGG COMPOUND", "type": "database", "url": "https://www.genome.jp/kegg/compound/", "description": "Collection of small molecules and biopolymers.", "tags": [ "compound" ], "tasks": [], "modalities": [ "Small Molecule" ], "organism": [], "api": false }, { "id": "kegg_drug", "name": "KEGG DRUG", "type": "database", "url": "https://www.genome.jp/kegg/drug/", "description": "Comprehensive, approved drug information.", "tags": [ "disease" ], "tasks": [], "modalities": [ "Disease" ], "organism": [], "api": false }, { "id": "kegg_pathway", "name": "KEGG PATHWAY", "type": "database", "url": "https://www.genome.jp/kegg/pathway.html", "description": "Collection of pathway maps.", "tags": [ "pathway" ], "tasks": [], "modalities": [ "Pathway" ], "organism": [], "api": false }, { "id": "kinase_inhibitor_bioactivity_data_kiba", "name": "Kinase Inhibitor Bioactivity Data (KIBA)", "type": "database", "url": "https://janeliascicomp.github.io/KIBA/", "description": "Integrated bioactivity scores for kinase inhibitors combining Ki, Kd, and IC50 measurements.", "tags": [ "chemical-protein-interaction", "interaction" ], "tasks": [], "modalities": [ "Protein", "Small Molecule" ], "organism": [], "api": false }, { "id": "lipid_maps", "name": "LIPID MAPS", "type": "database", "url": "https://www.lipidmaps.org/databases/lmsd/overview", "description": "Database of lipids.", "tags": [ "compound" ], "tasks": [], "modalities": [ "Small Molecule" ], "organism": [], "api": false }, { "id": "massbank", "name": "MassBank", "type": "database", "url": "http://www.massbank.jp/", "description": "Open source databases and tools for mass spectrometry reference spectra.", "tags": [ "mass-spectra" ], "tasks": [], "modalities": [ "Mass Spectra" ], "organism": [], "api": false }, { "id": "mgnify", "name": "MGnify", "type": "database", "url": "https://www.ebi.ac.uk/metagenomics/", "description": "Resource for metagenomic and metatranscriptomic data.", "tags": [ "genome" ], "tasks": [], "modalities": [ "Genomics" ], "organism": [], "api": false }, { "id": "mimic_iv", "name": "MIMIC-IV", "type": "database", "url": "https://mimic.mit.edu/", "description": "Freely accessible critical care database.", "tags": [ "clinical-trial" ], "tasks": [], "modalities": [ "Clinical" ], "organism": [], "api": false }, { "id": "mirbase", "name": "miRBase", "type": "database", "url": "https://www.mirbase.org/", "description": "Reference repository for microRNA gene annotations, sequences, and experimentally validated targets.", "tags": [ "gene-regulatory-network", "interaction" ], "tasks": [], "modalities": [ "Gene Expression" ], "organism": [], "api": false }, { "id": "mona_massbank_of_north_america", "name": "MoNA MassBank of North America", "type": "database", "url": "https://mona.fiehnlab.ucdavis.edu/", "description": "Meta-database of metabolite mass spectra, metadata, and associated compounds.", "tags": [ "mass-spectra" ], "tasks": [], "modalities": [ "Mass Spectra" ], "organism": [], "api": false }, { "id": "msigdb_molecular_signatures_database", "name": "MSigDB (Molecular Signatures Database)", "type": "database", "url": "https://www.gsea-msigdb.org/gsea/msigdb", "description": "Curated gene sets derived from pathways and biological processes.", "tags": [ "pathway" ], "tasks": [], "modalities": [ "Pathway" ], "organism": [], "api": false }, { "id": "nci60", "name": "NCI60", "type": "database", "url": "https://dtp.cancer.gov/discovery_development/nci-60/", "description": "Focuses on 60 cancer cell lines and many drugs.", "tags": [ "benchmarks-and-datasets", "drug-cell-line-response", "interaction" ], "tasks": [], "modalities": [ "Gene Expression", "Small Molecule" ], "organism": [], "api": false }, { "id": "nextprot", "name": "NeXtProt", "type": "database", "url": "https://www.nextprot.org/", "description": "Expert knowledge base on human proteins with deep functional annotation, complementary to UniProt.", "tags": [ "protein" ], "tasks": [], "modalities": [ "Protein" ], "organism": [], "api": false }, { "id": "oadb_observed_antibody_space_database", "name": "OADB (Observed Antibody Space Database)", "type": "database", "url": "http://opig.stats.ox.ac.uk/webapps/oas/", "description": "Database of antibody sequences from immune repertoire sequencing.", "tags": [ "protein" ], "tasks": [], "modalities": [ "Protein" ], "organism": [], "api": false }, { "id": "omim_online_mendelian_inheritance_in_man", "name": "OMIM (Online Mendelian Inheritance in Man)", "type": "database", "url": "https://www.omim.org/", "description": "Comprehensive database of human genes and genetic disorders.", "tags": [ "disease" ], "tasks": [], "modalities": [ "Disease" ], "organism": [], "api": false }, { "id": "omnipath", "name": "OmniPath", "type": "database", "url": "https://omnipathdb.org/", "description": "Comprehensive resource integrating protein interactions, signaling pathways, gene regulatory networks, and miRNA targets from over 100 databases.", "tags": [ "pathway" ], "tasks": [], "modalities": [ "Pathway" ], "organism": [], "api": false }, { "id": "oncokb", "name": "OncoKB", "type": "database", "url": "https://www.oncokb.org/", "description": "Precision oncology knowledge base of cancer genes, variants, and therapeutic implications.", "tags": [ "genome" ], "tasks": [], "modalities": [ "Genomics" ], "organism": [], "api": false }, { "id": "open_targets_platform", "name": "Open Targets Platform", "type": "database", "url": "https://platform.opentargets.org/", "description": "Systematic target identification and prioritization platform integrating genetics, genomics, and drug data for drug discovery.", "tags": [ "disease" ], "tasks": [], "modalities": [ "Disease" ], "organism": [], "api": false }, { "id": "pathwaycommons", "name": "PathwayCommons", "type": "database", "url": "https://www.pathwaycommons.org/", "description": "Database of pathways and interactions.", "tags": [ "pathway" ], "tasks": [], "modalities": [ "Pathway" ], "organism": [], "api": false }, { "id": "pdbbind", "name": "PDBBind", "type": "database", "url": "https://www.pdbbind-plus.org.cn/", "description": "Binding affinity data for biomolecular complexes.", "tags": [ "chemical-protein-interaction", "interaction" ], "tasks": [], "modalities": [ "Protein", "Small Molecule" ], "organism": [], "api": false }, { "id": "pfam", "name": "Pfam", "type": "database", "url": "https://www.ebi.ac.uk/interpro/entry/pfam/", "description": "Database of protein families described by multiple sequence alignments and hidden Markov models.", "tags": [ "protein" ], "tasks": [], "modalities": [ "Protein" ], "organism": [], "api": false }, { "id": "primekg", "name": "PrimeKG", "type": "database", "url": "https://github.com/mims-harvard/PrimeKG", "description": "Multi-modal precision medicine knowledge graph integrating clinical, genetic, and drug data.", "tags": [ "interaction", "knowledge-graph" ], "tasks": [], "modalities": [ "Knowledge Graph" ], "organism": [], "api": false }, { "id": "protein_data_bank_pdb", "name": "PROTEIN DATA BANK (PDB)", "type": "database", "url": "https://www.rcsb.org/", "description": "3D structures of proteins, nucleic acids, complexes.", "tags": [ "protein" ], "tasks": [], "modalities": [ "Protein" ], "organism": [], "api": false }, { "id": "pubchem", "name": "PubChem", "type": "database", "url": "https://pubchem.ncbi.nlm.nih.gov/", "description": "One of the largest chemical databases (compounds, genes, and proteins).", "tags": [ "compound" ], "tasks": [], "modalities": [ "Small Molecule" ], "organism": [], "api": false }, { "id": "rcsb_protein_data_bank", "name": "RCSB Protein Data Bank", "type": "database", "url": "https://www.rcsb.org/", "description": "Repository for structural data of biological molecules.", "tags": [ "protein" ], "tasks": [], "modalities": [ "Protein" ], "organism": [], "api": false }, { "id": "reactome", "name": "Reactome", "type": "database", "url": "https://reactome.org/", "description": "Expert-curated, peer-reviewed pathway database with detailed reaction mechanisms.", "tags": [ "pathway" ], "tasks": [], "modalities": [ "Pathway" ], "organism": [], "api": false }, { "id": "regnetwork", "name": "RegNetwork", "type": "database", "url": "http://www.regnetworkweb.org/", "description": "Database of gene regulatory networks covering transcription factor–target gene and miRNA–gene interaction data across multiple species.", "tags": [ "gene-regulatory-network", "interaction" ], "tasks": [], "modalities": [ "Gene Expression" ], "organism": [], "api": false }, { "id": "rfam", "name": "Rfam", "type": "database", "url": "https://rfam.org/", "description": "Database of RNA families with sequence alignments and consensus structures.", "tags": [ "genome" ], "tasks": [], "modalities": [ "Genomics" ], "organism": [], "api": false }, { "id": "rhea", "name": "Rhea", "type": "database", "url": "https://www.rhea-db.org/", "description": "Database of chemical reactions.", "tags": [ "compound" ], "tasks": [], "modalities": [ "Small Molecule" ], "organism": [], "api": false }, { "id": "roadmap_epigenomics", "name": "ROADMAP Epigenomics", "type": "database", "url": "http://www.roadmapepigenomics.org/", "description": "Reference epigenome maps for 111 primary human cell types and tissues, including histone modifications, chromatin accessibility, and DNA methylation.", "tags": [ "genome" ], "tasks": [], "modalities": [ "Genomics" ], "organism": [], "api": false }, { "id": "sabdab", "name": "SAbDab", "type": "database", "url": "https://opig.stats.ox.ac.uk/webapps/sabdab-sabpred/sabdab", "description": "Structural Antibody Database containing all antibody structures in the PDB.", "tags": [ "protein" ], "tasks": [], "modalities": [ "Protein" ], "organism": [], "api": false }, { "id": "signor_2_0", "name": "SIGNOR 2.0", "type": "database", "url": "https://signor.uniroma2.it/", "description": "Database of causal signaling interactions and pathways, with signed and directed relationships between proteins.", "tags": [ "pathway" ], "tasks": [], "modalities": [ "Pathway" ], "organism": [], "api": false }, { "id": "single_cell_expression_atlas", "name": "Single Cell Expression Atlas", "type": "database", "url": "https://www.ebi.ac.uk/gxa/sc/home", "description": "Public database for single-cell RNA.", "tags": [ "scrna" ], "tasks": [], "modalities": [ "Single Cell" ], "organism": [], "api": false }, { "id": "single_cell_portal", "name": "Single Cell PORTAL", "type": "database", "url": "https://singlecell.broadinstitute.org/single_cell", "description": "Public database for single-cell RNA.", "tags": [ "scrna" ], "tasks": [], "modalities": [ "Single Cell" ], "organism": [], "api": false }, { "id": "snap", "name": "SNAP", "type": "database", "url": "https://snap.stanford.edu/biodata/datasets/10002/10002-ChG-Miner.html", "description": "Dataset of drug-gene interactions.", "tags": [ "drug-gene-interaction", "interaction" ], "tasks": [], "modalities": [ "Gene", "Small Molecule" ], "organism": [], "api": false }, { "id": "stitch", "name": "STITCH", "type": "database", "url": "http://stitch.embl.de/", "description": "Chemical-protein interactions.", "tags": [ "chemical-protein-interaction", "interaction" ], "tasks": [], "modalities": [ "Protein", "Small Molecule" ], "organism": [], "api": false }, { "id": "string", "name": "STRING", "type": "database", "url": "https://string-db.org/", "description": "PPI networks for multiple organisms.", "tags": [ "interaction", "protein-protein-interaction" ], "tasks": [], "modalities": [ "knowledge-graph", "proteomics" ], "organism": [], "api": false, "entities": [ "protein" ], "documentation": "https://string-db.org/help/api/", "last_checked": "2026-08-08", "metadata_sources": [ "https://string-db.org/", "https://string-db.org/help/api/" ] }, { "id": "the_genotype_tissue_expression_gtex", "name": "The Genotype-Tissue Expression (GTEx)", "type": "database", "url": "https://gtexportal.org/home/", "description": "Human gene expression and regulation resource.", "tags": [ "genome" ], "tasks": [], "modalities": [ "Genomics" ], "organism": [], "api": false }, { "id": "the_human_protein_atlas", "name": "THE HUMAN PROTEIN ATLAS", "type": "database", "url": "https://www.proteinatlas.org/", "description": "Comprehensive human protein database (cells, tissues, organs).", "tags": [ "protein" ], "tasks": [], "modalities": [ "Protein" ], "organism": [], "api": false }, { "id": "therapeutic_target_database", "name": "Therapeutic Target Database", "type": "database", "url": "https://idrblab.net/ttd/full-data-download", "description": "Drug-target, target-disease, and drug-disease datasets.", "tags": [ "compound" ], "tasks": [], "modalities": [ "Small Molecule" ], "organism": [], "api": false }, { "id": "trrust_v2", "name": "TRRUST v2", "type": "database", "url": "https://www.grnpedia.org/trrust/", "description": "Manually curated database of human and mouse transcriptional regulatory interactions between transcription factors and their target genes, expanded with literature-derived evidence.", "tags": [ "gene-regulatory-network", "interaction" ], "tasks": [], "modalities": [ "Gene Expression" ], "organism": [], "api": false }, { "id": "ucsc_genome_browser", "name": "UCSC Genome Browser", "type": "database", "url": "https://genome.ucsc.edu/", "description": "UCSC's genome browser.", "tags": [ "genome" ], "tasks": [], "modalities": [ "Genomics" ], "organism": [], "api": false }, { "id": "uniclust", "name": "Uniclust", "type": "database", "url": "https://uniclust.mmseqs.com/", "description": "Clustered protein sequence databases.", "tags": [ "protein" ], "tasks": [], "modalities": [ "Protein" ], "organism": [], "api": false }, { "id": "uniprot", "name": "UniProt", "type": "database", "url": "https://www.uniprot.org/", "description": "Functional information on proteins.", "tags": [ "protein" ], "tasks": [], "modalities": [ "protein-sequence" ], "organism": [], "api": false, "entities": [ "protein" ], "documentation": "https://www.uniprot.org/", "last_checked": "2026-08-08", "metadata_sources": [ "https://www.uniprot.org/" ] }, { "id": "uniref", "name": "UniRef", "type": "database", "url": "https://www.uniprot.org/uniref/", "description": "Non-redundant sequence database clustering UniProtKB entries at multiple sequence identity thresholds.", "tags": [ "protein" ], "tasks": [], "modalities": [ "Protein" ], "organism": [], "api": false }, { "id": "wikipathways", "name": "WikiPathways", "type": "database", "url": "https://wikipathways.org/", "description": "Database of biological pathways.", "tags": [ "pathway" ], "tasks": [], "modalities": [ "Pathway" ], "organism": [], "api": false }, { "id": "zinc_ligand_discovery_database", "name": "ZINC ligand discovery database", "type": "database", "url": "https://zinc.docking.org/", "description": "Free database of commercially-available compounds for virtual screening.", "tags": [ "compound" ], "tasks": [], "modalities": [ "Small Molecule" ], "organism": [], "api": false }, { "id": "aestetik", "name": "AESTETIK", "type": "model", "url": "https://github.com/ratschlab/aestetik", "description": "Autoencoder for spatial transcriptomics representation learning using topology and histology image knowledge.", "tags": [ "foundation-models", "single-cell-foundation-models", "spatial-foundation-models" ], "tasks": [ "representation-learning" ], "modalities": [ "histopathology", "spatial-transcriptomics" ], "organism": [], "api": false, "entities": [ "cell", "gene", "tissue" ], "methods": [ "autoencoder" ], "github": "https://github.com/ratschlab/aestetik", "last_checked": "2026-08-08", "metadata_sources": [ "https://github.com/ratschlab/aestetik" ] }, { "id": "ai4chem_chemllm_7b_chat", "name": "AI4Chem/ChemLLM-7B-Chat", "type": "model", "url": "https://huggingface.co/AI4Chem/ChemLLM-7B-Chat", "description": "LLM for chemical & molecular science.", "tags": [ "llm-for-biology" ], "tasks": [ "Language Modeling" ], "modalities": [ "Text" ], "organism": [], "api": false }, { "id": "alphafold3", "name": "AlphaFold3", "type": "model", "url": "https://github.com/google-deepmind/alphafold3", "description": "Predicts structures of proteins, nucleic acids, small molecules, and their complexes.", "tags": [ "foundation-models", "protein-foundation-models", "protein-structure-prediction-and-design" ], "tasks": [ "structure-prediction" ], "modalities": [ "molecular-structure", "protein-sequence" ], "organism": [], "api": false, "entities": [ "molecule", "protein", "protein-complex" ], "methods": [ "diffusion" ], "github": "https://github.com/google-deepmind/alphafold3", "last_checked": "2026-08-08", "metadata_sources": [ "https://github.com/google-deepmind/alphafold3" ] }, { "id": "ankh", "name": "Ankh", "type": "model", "url": "https://github.com/agemagician/Ankh", "description": "Efficient protein language model optimized for downstream prediction tasks including secondary structure, localization, and function annotation.", "tags": [ "foundation-models", "pre-trained-embedding", "protein-foundation-models" ], "tasks": [ "Foundation Model" ], "modalities": [ "Protein" ], "organism": [], "api": false }, { "id": "babel", "name": "BABEL", "type": "model", "url": "https://github.com/wukevin/babel", "description": "Cross-modality translation model enabling prediction between scRNA-seq and scATAC-seq profiles without requiring paired single-cell measurements.", "tags": [ "foundation-models", "multi-omics-foundation-models", "single-cell-foundation-models" ], "tasks": [ "Foundation Model" ], "modalities": [ "Multi-Omics", "Single Cell" ], "organism": [], "api": false }, { "id": "basenji", "name": "Basenji", "type": "model", "url": "https://github.com/calico/basenji", "description": "Sequential regulatory activity prediction from DNA sequences.", "tags": [ "foundation-models", "genomics-foundation-models" ], "tasks": [ "Foundation Model" ], "modalities": [ "Genomics" ], "organism": [], "api": false }, { "id": "biogpt", "name": "BioGPT", "type": "model", "url": "https://github.com/microsoft/BioGPT", "description": "LLM for biomedical text generation.", "tags": [ "llm-for-biology" ], "tasks": [ "Language Modeling" ], "modalities": [ "Text" ], "organism": [], "api": false }, { "id": "biomedclip", "name": "BiomedCLIP", "type": "model", "url": "https://huggingface.co/microsoft/BiomedCLIP-PubMedBERT_256-vit_g_14", "description": "CLIP-based vision-language foundation model for biomedical images and text trained on PubMed figure–caption pairs.", "tags": [ "foundation-models", "multi-modal-foundation-models" ], "tasks": [ "Foundation Model" ], "modalities": [ "Multi-Modal" ], "organism": [], "api": false }, { "id": "biomedlm", "name": "BioMedLM", "type": "model", "url": "https://huggingface.co/stanford-crfm/BioMedLM", "description": "2.7B parameter GPT-2-style language model trained exclusively on biomedical literature from PubMed for biomedical question answering and text generation.", "tags": [ "llm-for-biology" ], "tasks": [ "Language Modeling" ], "modalities": [ "Text" ], "organism": [], "api": false }, { "id": "boltz_1", "name": "Boltz-1", "type": "model", "url": "https://github.com/jwohlwend/boltz", "description": "Open-source all-atom biomolecular structure prediction model for proteins, nucleic acids, small molecules, and their complexes achieving AlphaFold3-level accuracy.", "tags": [ "foundation-models", "protein-foundation-models", "protein-structure-prediction-and-design" ], "tasks": [ "Foundation Model", "Protein Structure Prediction" ], "modalities": [ "Protein" ], "organism": [], "api": false }, { "id": "borzoi", "name": "Borzoi", "type": "model", "url": "https://github.com/calico/borzoi", "description": "Extended successor to Enformer for predicting RNA-seq coverage from long genomic sequence windows (524 kb) with improved resolution.", "tags": [ "foundation-models", "genomics-foundation-models" ], "tasks": [ "Foundation Model" ], "modalities": [ "Genomics" ], "organism": [], "api": false }, { "id": "bulkformer", "name": "BulkFormer", "type": "model", "url": "https://github.com/KangBoming/BulkFormer", "description": "Foundation model for bulk RNA-seq data; learns general transcriptomic representations.", "tags": [ "foundation-models", "single-cell-foundation-models", "transcriptomics-foundation-models" ], "tasks": [ "Foundation Model" ], "modalities": [ "Single Cell", "Transcriptomics" ], "organism": [], "api": false }, { "id": "caduceus", "name": "Caduceus", "type": "model", "url": "https://github.com/kuleshov-group/caduceus", "description": "Bidirectional equivariant long-range DNA sequence model based on Mamba.", "tags": [ "foundation-models", "genomics-foundation-models" ], "tasks": [ "Foundation Model" ], "modalities": [ "Genomics" ], "organism": [], "api": false }, { "id": "cancerfoundation", "name": "CancerFoundation", "type": "model", "url": "https://github.com/BoevaLab/CancerFoundation", "description": "Single-cell RNA-seq foundation model trained exclusively on a curated dataset of malignant cells to learn cancer-specific embeddings.", "tags": [ "foundation-models", "single-cell-foundation-models", "transcriptomics-foundation-models" ], "tasks": [ "Foundation Model" ], "modalities": [ "Single Cell", "Transcriptomics" ], "organism": [], "api": false }, { "id": "cassia", "name": "CASSIA", "type": "model", "url": "https://github.com/ElliotXie/CASSIA", "description": "Multi-agent LLM for reference-free, interpretable cell-type annotation of single-cell RNA-seq data, with dedicated annotation, validation, scoring, and reporting agents.", "tags": [ "llm-for-biology" ], "tasks": [ "Language Modeling" ], "modalities": [ "Text" ], "organism": [], "api": false }, { "id": "cellot", "name": "CellOT", "type": "model", "url": "https://github.com/bunnech/cellot", "description": "Neural optimal transport framework for predicting single-cell responses to drug and genetic perturbations.", "tags": [ "drug-discovery", "drug-perturbation" ], "tasks": [ "Drug Discovery", "Drug Perturbation" ], "modalities": [ "Small Molecule" ], "organism": [], "api": false }, { "id": "cellplm", "name": "CellPLM", "type": "model", "url": "https://github.com/OmicsML/CellPLM", "description": "Cell pre-trained language model with inter-cell transformer architecture for diverse single-cell analysis tasks.", "tags": [ "foundation-models", "single-cell-foundation-models", "transcriptomics-foundation-models" ], "tasks": [ "foundation-model-pretraining", "representation-learning" ], "modalities": [ "single-cell-rna-seq", "transcriptomics" ], "organism": [], "api": false, "entities": [ "cell", "gene" ], "methods": [ "self-supervised-learning", "transformer" ], "year": 2023, "github": "https://github.com/OmicsML/CellPLM", "paper": "https://www.biorxiv.org/content/10.1101/2023.10.03.560734v1", "last_checked": "2026-08-08", "metadata_sources": [ "https://github.com/OmicsML/CellPLM", "https://www.biorxiv.org/content/10.1101/2023.10.03.560734v1" ] }, { "id": "chai_1", "name": "Chai-1", "type": "model", "url": "https://github.com/chaidiscovery/chai-lab", "description": "Unified molecular structure prediction model covering proteins, nucleic acids, small molecules, and complexes.", "tags": [ "foundation-models", "protein-foundation-models", "protein-structure-prediction-and-design" ], "tasks": [ "Foundation Model", "Protein Structure Prediction" ], "modalities": [ "Protein" ], "organism": [], "api": false }, { "id": "chatdrug", "name": "ChatDrug", "type": "model", "url": "https://github.com/chao1224/ChatDrug", "description": "LLM-based conversational pipeline for drug discovery, using natural language prompts for iterative drug editing and optimization.", "tags": [ "llm-for-biology" ], "tasks": [ "Language Modeling" ], "modalities": [ "Text" ], "organism": [], "api": false }, { "id": "chemberta_2", "name": "ChemBERTa-2", "type": "model", "url": "https://github.com/seyonechithrananda/bert-loves-chemistry", "description": "RoBERTa-based molecular language model pretrained on SMILES for small-molecule representation learning.", "tags": [ "compound-embedding", "compound-foundation-models", "foundation-models" ], "tasks": [ "representation-learning" ], "modalities": [ "chemical-structure" ], "organism": [], "api": false, "entities": [ "molecule" ], "methods": [ "language-model", "self-supervised-learning", "transformer" ], "github": "https://github.com/seyonechithrananda/bert-loves-chemistry", "last_checked": "2026-08-08", "metadata_sources": [ "https://github.com/seyonechithrananda/bert-loves-chemistry" ] }, { "id": "chemcpa", "name": "chemCPA", "type": "model", "url": "https://github.com/theislab/chemCPA", "description": "Compositional perturbation autoencoder for predicting single-cell transcriptional responses to unseen drug perturbations and dose combinations.", "tags": [ "drug-discovery", "drug-perturbation" ], "tasks": [ "Drug Discovery", "Drug Perturbation" ], "modalities": [ "Small Molecule" ], "organism": [], "api": false }, { "id": "chief", "name": "CHIEF", "type": "model", "url": "https://github.com/hms-dbmi/CHIEF", "description": "Clinical Histopathology Imaging Evaluation Foundation model integrating histology images and clinical context for pan-cancer analysis.", "tags": [ "foundation-models", "multi-modal-foundation-models" ], "tasks": [ "Foundation Model" ], "modalities": [ "Multi-Modal" ], "organism": [], "api": false }, { "id": "clawbio", "name": "ClawBio", "type": "model", "url": "https://github.com/ClawBio/ClawBio", "description": "Bioinformatics-native AI agent skill library with local-first pharmacogenomics, ancestry PCA, semantic similarity, nutrigenomics, and metagenomics skills.", "tags": [ "llm-for-biology" ], "tasks": [ "Language Modeling" ], "modalities": [ "Text" ], "organism": [], "api": false }, { "id": "cmonge", "name": "CMonge", "type": "model", "url": "https://github.com/AI4SCR/conditional-monge-gap", "description": "Conditional optimal transport model for generalizable single-cell perturbation response prediction across drugs and doses.", "tags": [ "drug-discovery", "drug-perturbation" ], "tasks": [ "Drug Discovery", "Drug Perturbation" ], "modalities": [ "Small Molecule" ], "organism": [], "api": false }, { "id": "concerto", "name": "Concerto", "type": "model", "url": "https://github.com/melobio/Concerto-reproducibility", "description": "Contrastive self-supervised learning framework for single-cell multimodal data integration, batch correction, and reference-query mapping.", "tags": [ "foundation-models", "multi-omics-foundation-models", "single-cell-foundation-models" ], "tasks": [ "Foundation Model" ], "modalities": [ "Multi-Omics", "Single Cell" ], "organism": [], "api": false }, { "id": "conch", "name": "CONCH", "type": "model", "url": "https://github.com/mahmoodlab/CONCH", "description": "Vision-language foundation model for computational pathology trained with contrastive captioning on pathology image–text pairs.", "tags": [ "foundation-models", "single-cell-foundation-models", "spatial-foundation-models" ], "tasks": [ "foundation-model-pretraining", "representation-learning" ], "modalities": [ "histopathology", "imaging" ], "organism": [], "api": false, "entities": [ "tissue" ], "methods": [ "contrastive-learning", "transformer" ], "github": "https://github.com/mahmoodlab/CONCH", "last_checked": "2026-08-08", "metadata_sources": [ "https://github.com/mahmoodlab/CONCH" ] }, { "id": "cyclecdr", "name": "cycleCDR", "type": "model", "url": "https://github.com/hliulab/cycleCDR", "description": "Interpretable cycle-consistency framework for modeling cellular responses to drug perturbations.", "tags": [ "drug-discovery", "drug-perturbation" ], "tasks": [ "Drug Discovery", "Drug Perturbation" ], "modalities": [ "Small Molecule" ], "organism": [], "api": false }, { "id": "deepaeg", "name": "DeepAEG", "type": "model", "url": "https://github.com/zhejiangzhuque/DeepAEG", "description": "GNN embedding + attention mechanism.", "tags": [ "drug-discovery", "drug-response-prediction" ], "tasks": [ "Drug Discovery", "Drug Response Prediction" ], "modalities": [ "Small Molecule" ], "organism": [], "api": false }, { "id": "deepdsc", "name": "DeepDSC", "type": "model", "url": "https://ieeexplore-ieee-org.ezp2.lib.umn.edu/stamp/stamp.jsp?tp=&arnumber=8723620&tag=1", "description": "Autoencoder + fully connected NN.", "tags": [ "drug-discovery", "drug-response-prediction" ], "tasks": [ "Drug Discovery", "Drug Response Prediction" ], "modalities": [ "Small Molecule" ], "organism": [], "api": false }, { "id": "deepdta", "name": "DeepDTA", "type": "model", "url": "https://github.com/hkmztrk/DeepDTA", "description": "Deep learning model using CNNs on protein sequences and drug SMILES.", "tags": [ "drug-discovery", "drug-target-interaction" ], "tasks": [ "Drug Discovery", "Drug Target Interaction" ], "modalities": [ "Protein", "Small Molecule" ], "organism": [], "api": false }, { "id": "deeppurpose", "name": "DeepPurpose", "type": "model", "url": "https://github.com/kexinhuang12345/DeepPurpose", "description": "Deep learning library for drug repurposing.", "tags": [ "drug-discovery", "drug-repurposing" ], "tasks": [ "Drug Discovery", "Drug Repurposing" ], "modalities": [ "Small Molecule" ], "organism": [], "api": false }, { "id": "deepsea", "name": "DeepSEA", "type": "model", "url": "http://deepsea.princeton.edu/", "description": "Deep learning framework for predicting chromatin effects of sequence alterations with single-nucleotide sensitivity across thousands of chromatin features.", "tags": [ "foundation-models", "genomics-foundation-models" ], "tasks": [ "Foundation Model" ], "modalities": [ "Genomics" ], "organism": [], "api": false }, { "id": "deepspot", "name": "DeepSpot", "type": "model", "url": "https://github.com/ratschlab/DeepSpot", "description": "Deep learning model predicting spatial transcriptomics from H&E images at spot and single-cell resolution.", "tags": [ "foundation-models", "single-cell-foundation-models", "spatial-foundation-models" ], "tasks": [ "regression" ], "modalities": [ "histopathology", "spatial-transcriptomics" ], "organism": [], "api": false, "entities": [ "gene", "tissue" ], "github": "https://github.com/ratschlab/DeepSpot", "last_checked": "2026-08-08", "metadata_sources": [ "https://github.com/ratschlab/DeepSpot" ] }, { "id": "deepspot_m", "name": "DeepSpot-M", "type": "model", "url": "https://github.com/ratschlab/DeepSpotM", "description": "Multimodal foundation model for transcriptome-wide virtual spatial transcriptomics from histology.", "tags": [ "foundation-models", "single-cell-foundation-models", "spatial-foundation-models" ], "tasks": [ "foundation-model-pretraining", "regression" ], "modalities": [ "histopathology", "spatial-transcriptomics", "transcriptomics" ], "organism": [], "api": false, "entities": [ "gene", "tissue" ], "github": "https://github.com/ratschlab/DeepSpotM", "last_checked": "2026-08-08", "metadata_sources": [ "https://github.com/ratschlab/DeepSpotM" ] }, { "id": "deepspot2cell", "name": "DeepSpot2Cell", "type": "model", "url": "https://github.com/ratschlab/DeepSpot2Cell", "description": "Predicts virtual single-cell spatial transcriptomics from H&E using spot-level supervision (NeurIPS 2025 Imageomics).", "tags": [ "foundation-models", "single-cell-foundation-models", "spatial-foundation-models" ], "tasks": [ "regression" ], "modalities": [ "histopathology", "spatial-transcriptomics" ], "organism": [], "api": false, "entities": [ "cell", "gene", "tissue" ], "github": "https://github.com/ratschlab/DeepSpot2Cell", "last_checked": "2026-08-08", "metadata_sources": [ "https://github.com/ratschlab/DeepSpot2Cell" ] }, { "id": "dgdrp", "name": "DGDRP", "type": "model", "url": "https://github.com/minwoopak/heteronet", "description": "Multi-view embedding neural network.", "tags": [ "drug-discovery", "drug-response-prediction" ], "tasks": [ "Drug Discovery", "Drug Response Prediction" ], "modalities": [ "Small Molecule" ], "organism": [], "api": false }, { "id": "diffdock", "name": "DiffDock", "type": "model", "url": "https://github.com/gcorso/DiffDock", "description": "Diffusion generative model for molecular docking, predicting the binding pose of small molecules to protein targets.", "tags": [ "drug-discovery", "molecular-generation" ], "tasks": [ "docking" ], "modalities": [ "molecular-structure" ], "organism": [], "api": false, "entities": [ "molecule", "protein" ], "methods": [ "diffusion", "geometric-deep-learning" ], "year": 2023, "github": "https://github.com/gcorso/DiffDock", "paper": "https://openreview.net/forum?id=kKF8_K-mBbS", "last_checked": "2026-08-08", "metadata_sources": [ "https://github.com/gcorso/DiffDock", "https://openreview.net/forum?id=kKF8_K-mBbS" ] }, { "id": "diffsbdd", "name": "DiffSBDD", "type": "model", "url": "https://github.com/arneschneuing/DiffSBDD", "description": "Equivariant diffusion model for structure-based drug design that generates molecules and binding conformations for protein targets.", "tags": [ "drug-discovery", "molecular-generation" ], "tasks": [ "Drug Discovery", "Molecular Generation" ], "modalities": [ "Small Molecule" ], "organism": [], "api": false }, { "id": "dnabert", "name": "DNABERT", "type": "model", "url": "https://github.com/jerryji1993/DNABERT", "description": "Pre-trained bidirectional encoder for DNA sequence analysis.", "tags": [ "foundation-models", "genomics-foundation-models" ], "tasks": [ "Foundation Model" ], "modalities": [ "Genomics" ], "organism": [], "api": false }, { "id": "dnabert_2", "name": "DNABERT-2", "type": "model", "url": "https://github.com/Zhihan1996/DNABERT_2", "description": "Improved genome foundation model with efficient tokenization.", "tags": [ "foundation-models", "genomics-foundation-models" ], "tasks": [ "Foundation Model" ], "modalities": [ "Genomics" ], "organism": [], "api": false }, { "id": "drgat", "name": "drGAT", "type": "model", "url": "https://github.com/inoue0426/drGAT", "description": "Attention-based model for drug response prediction with gene explainability.", "tags": [ "drug-discovery", "drug-response-prediction" ], "tasks": [ "Drug Discovery", "Drug Response Prediction" ], "modalities": [ "Small Molecule" ], "organism": [], "api": false }, { "id": "drugban", "name": "DrugBAN", "type": "model", "url": "https://github.com/peizhenbai/DrugBAN", "description": "Bilinear attention network for interpretable DTI prediction.", "tags": [ "drug-discovery", "drug-target-interaction" ], "tasks": [ "Drug Discovery", "Drug Target Interaction" ], "modalities": [ "Protein", "Small Molecule" ], "organism": [], "api": false }, { "id": "druml", "name": "DRUML", "type": "model", "url": "https://github.com/CutillasLab/DRUMLR", "description": "Ensemble machine learning framework combining standard ML with deep learning to systematically rank anti-cancer drugs from proteomics and RNA-seq data.", "tags": [ "drug-discovery", "drug-response-prediction" ], "tasks": [ "Drug Discovery", "Drug Response Prediction" ], "modalities": [ "Small Molecule" ], "organism": [], "api": false }, { "id": "dtinet", "name": "DTINet", "type": "model", "url": "https://github.com/luoyunan/DTINet", "description": "Network-based framework integrating heterogeneous biological data for DTI prediction.", "tags": [ "drug-discovery", "drug-target-interaction" ], "tasks": [ "Drug Discovery", "Drug Target Interaction" ], "modalities": [ "Protein", "Small Molecule" ], "organism": [], "api": false }, { "id": "enformer", "name": "Enformer", "type": "model", "url": "https://github.com/deepmind/deepmind-research/tree/master/enformer", "description": "Transformer model predicting gene expression from DNA sequence.", "tags": [ "foundation-models", "genomics-foundation-models" ], "tasks": [ "Foundation Model" ], "modalities": [ "Genomics" ], "organism": [], "api": false }, { "id": "esm3", "name": "ESM3", "type": "model", "url": "https://github.com/evolutionaryscale/esm", "description": "Multimodal protein language model that jointly reasons over sequence, structure, and function for generative protein design and engineering.", "tags": [ "foundation-models", "protein-foundation-models", "protein-structure-prediction-and-design" ], "tasks": [ "protein-sequence-design", "representation-learning" ], "modalities": [ "molecular-structure", "protein-sequence" ], "organism": [], "api": false, "entities": [ "protein" ], "methods": [ "generative-model", "language-model", "transformer" ], "github": "https://github.com/evolutionaryscale/esm", "last_checked": "2026-08-08", "metadata_sources": [ "https://github.com/evolutionaryscale/esm" ] }, { "id": "esmfold", "name": "ESMFold", "type": "model", "url": "https://github.com/facebookresearch/esm", "description": "Fast protein structure prediction using language model embeddings.", "tags": [ "foundation-models", "protein-foundation-models", "protein-structure-prediction-and-design" ], "tasks": [ "representation-learning", "structure-prediction" ], "modalities": [ "molecular-structure", "protein-sequence" ], "organism": [], "api": false, "entities": [ "protein" ], "methods": [ "language-model", "transformer" ], "year": 2023, "github": "https://github.com/facebookresearch/esm", "last_checked": "2026-08-08", "metadata_sources": [ "https://github.com/facebookresearch/esm" ] }, { "id": "evo", "name": "Evo", "type": "model", "url": "https://github.com/evo-design/evo", "description": "Long-context genomic foundation model (up to 1M tokens).", "tags": [ "foundation-models", "genomics-foundation-models" ], "tasks": [ "Foundation Model" ], "modalities": [ "Genomics" ], "organism": [], "api": false }, { "id": "evodiff", "name": "EvoDiff", "type": "model", "url": "https://github.com/microsoft/evodiff", "description": "Discrete diffusion framework for protein sequence generation trained on evolutionary-scale data, supporting unconditional generation, disordered region design, and functional motif scaffolding. [ [paper-2023](https://www.biorxiv.org/content/10.1101/2023.09.11.556673v1) ]", "tags": [ "foundation-models", "protein-foundation-models", "protein-structure-prediction-and-design" ], "tasks": [ "Foundation Model", "Protein Structure Prediction" ], "modalities": [ "Protein" ], "organism": [], "api": false }, { "id": "evolutionary_scale_modeling_esm", "name": "Evolutionary Scale Modeling (ESM)", "type": "model", "url": "https://github.com/facebookresearch/esm", "description": "Protein embeddings.", "tags": [ "foundation-models", "pre-trained-embedding", "protein-foundation-models" ], "tasks": [ "representation-learning" ], "modalities": [ "protein-sequence" ], "organism": [], "api": false, "entities": [ "protein" ], "methods": [ "language-model", "self-supervised-learning", "transformer" ], "github": "https://github.com/facebookresearch/esm", "last_checked": "2026-08-08", "metadata_sources": [ "https://github.com/facebookresearch/esm" ] }, { "id": "gears", "name": "GEARS", "type": "model", "url": "https://github.com/snap-stanford/GEARS", "description": "Graph-based model for predicting transcriptional responses to single and combinatorial genetic perturbations using biological priors.", "tags": [ "foundation-models", "single-cell-foundation-models", "transcriptomics-foundation-models" ], "tasks": [ "Foundation Model" ], "modalities": [ "Single Cell", "Transcriptomics" ], "organism": [], "api": false }, { "id": "genecompass", "name": "GeneCompass", "type": "model", "url": "https://github.com/xCompass-AI/GeneCompass", "description": "Large-scale foundation model integrating DNA regulatory sequences and single-cell transcriptomics from 120M+ cells across multiple species for gene regulation prediction.", "tags": [ "foundation-models", "multi-omics-foundation-models", "single-cell-foundation-models" ], "tasks": [ "foundation-model-pretraining", "representation-learning" ], "modalities": [ "single-cell-rna-seq", "transcriptomics" ], "organism": [], "api": false, "entities": [ "cell", "gene" ], "methods": [ "self-supervised-learning", "transformer" ], "year": 2024, "github": "https://github.com/xCompass-AI/GeneCompass", "paper": "https://www.nature.com/articles/s41422-024-01034-y", "last_checked": "2026-08-08", "metadata_sources": [ "https://github.com/xCompass-AI/GeneCompass", "https://www.nature.com/articles/s41422-024-01034-y" ] }, { "id": "geneformer", "name": "Geneformer", "type": "model", "url": "https://huggingface.co/ctheodoris/Geneformer", "description": "Context-aware, attention-based deep learning model pretrained on a large corpus of single-cell transcriptomes.", "tags": [ "foundation-models", "single-cell-foundation-models", "transcriptomics-foundation-models" ], "tasks": [ "classification", "foundation-model-pretraining", "perturbation-prediction", "representation-learning" ], "modalities": [ "single-cell-rna-seq", "transcriptomics" ], "organism": [], "api": false, "entities": [ "cell", "gene" ], "methods": [ "self-supervised-learning", "transformer" ], "year": 2023, "documentation": "https://geneformer.readthedocs.io/", "paper": "https://www.nature.com/articles/s41586-023-06139-9", "last_checked": "2026-08-08", "metadata_sources": [ "https://huggingface.co/ctheodoris/Geneformer", "https://www.nature.com/articles/s41586-023-06139-9" ] }, { "id": "genegpt", "name": "GeneGPT", "type": "model", "url": "https://github.com/ncbi/GeneGPT", "description": "LLM for biomedical information, integrated with various APIs.", "tags": [ "llm-for-biology" ], "tasks": [ "Language Modeling" ], "modalities": [ "Text" ], "organism": [], "api": false }, { "id": "genept", "name": "GenePT", "type": "model", "url": "https://github.com/yiqunchen/GenePT", "description": "Foundation LLM for single-cell data.", "tags": [ "llm-for-biology" ], "tasks": [ "batch-correction", "classification", "representation-learning" ], "modalities": [ "single-cell-rna-seq", "transcriptomics" ], "organism": [], "api": false, "entities": [ "cell", "gene" ], "methods": [ "language-model" ], "year": 2023, "github": "https://github.com/yiqunchen/GenePT", "paper": "https://www.biorxiv.org/content/10.1101/2023.10.16.562533v2", "last_checked": "2026-08-08", "metadata_sources": [ "https://github.com/yiqunchen/GenePT", "https://www.biorxiv.org/content/10.1101/2023.10.16.562533v2" ] }, { "id": "gigapath", "name": "GigaPath", "type": "model", "url": "https://github.com/prov-gigapath/prov-gigapath", "description": "Slide-level digital pathology foundation model pretrained on 1.3 billion pathology image tokens from whole-slide images.", "tags": [ "foundation-models", "single-cell-foundation-models", "spatial-foundation-models" ], "tasks": [ "foundation-model-pretraining", "representation-learning" ], "modalities": [ "histopathology", "imaging" ], "organism": [], "api": false, "entities": [ "tissue" ], "methods": [ "self-supervised-learning", "transformer" ], "github": "https://github.com/prov-gigapath/prov-gigapath", "last_checked": "2026-08-08", "metadata_sources": [ "https://github.com/prov-gigapath/prov-gigapath" ] }, { "id": "glue", "name": "GLUE", "type": "model", "url": "https://github.com/gao-lab/GLUE", "description": "Graph-Linked Unified Embedding framework for unpaired single-cell multi-omics data integration across RNA, ATAC, methylation, and protein modalities.", "tags": [ "foundation-models", "multi-omics-foundation-models", "single-cell-foundation-models" ], "tasks": [ "Foundation Model" ], "modalities": [ "Multi-Omics", "Single Cell" ], "organism": [], "api": false }, { "id": "gpn_genomic_pre_trained_network", "name": "GPN (Genomic Pre-trained Network)", "type": "model", "url": "https://github.com/songlab-cal/gpn", "description": "Masked language model for DNA sequences enabling zero-shot variant effect prediction without requiring functional annotations.", "tags": [ "foundation-models", "genomics-foundation-models" ], "tasks": [ "Foundation Model" ], "modalities": [ "Genomics" ], "organism": [], "api": false }, { "id": "graphdta", "name": "GraphDTA", "type": "model", "url": "https://github.com/thinng/GraphDTA", "description": "Graph neural network–based DTI prediction using molecular graphs.", "tags": [ "drug-discovery", "drug-target-interaction" ], "tasks": [ "Drug Discovery", "Drug Target Interaction" ], "modalities": [ "Protein", "Small Molecule" ], "organism": [], "api": false }, { "id": "grover", "name": "GROVER", "type": "model", "url": "https://github.com/tencent-ailab/grover", "description": "Self-supervised graph transformer for large-scale molecular representation learning from unlabeled compounds.", "tags": [ "compound-embedding", "compound-foundation-models", "foundation-models" ], "tasks": [ "representation-learning" ], "modalities": [ "chemical-structure" ], "organism": [], "api": false, "entities": [ "molecule" ], "methods": [ "graph-neural-network", "self-supervised-learning", "transformer" ], "github": "https://github.com/tencent-ailab/grover", "last_checked": "2026-08-08", "metadata_sources": [ "https://github.com/tencent-ailab/grover" ] }, { "id": "hidra", "name": "HiDRA", "type": "model", "url": "https://github.com/bsml320/HiDRA", "description": "Hierarchical network model incorporating gene and pathway-level information for cancer drug response prediction.", "tags": [ "drug-discovery", "drug-response-prediction" ], "tasks": [ "Drug Discovery", "Drug Response Prediction" ], "modalities": [ "Small Molecule" ], "organism": [], "api": false }, { "id": "hyenadna", "name": "HyenaDNA", "type": "model", "url": "https://github.com/HazyResearch/hyena-dna", "description": "Long-range genomic foundation model handling sequences up to 1M tokens with sub-quadratic attention.", "tags": [ "foundation-models", "genomics-foundation-models" ], "tasks": [ "Foundation Model" ], "modalities": [ "Genomics" ], "organism": [], "api": false }, { "id": "jamie", "name": "JAMIE", "type": "model", "url": "https://github.com/Oafish1/JAMIE", "description": "Joint variational autoencoder for multimodal single-cell data imputation and embedding.", "tags": [ "foundation-models", "multi-omics-foundation-models", "single-cell-foundation-models" ], "tasks": [ "Foundation Model" ], "modalities": [ "Multi-Omics", "Single Cell" ], "organism": [], "api": false }, { "id": "jtvae", "name": "JTVAE", "type": "model", "url": "https://github.com/wengong-jin/icml18-jtnn", "description": "Junction tree variational autoencoder for molecular graph generation that guarantees chemical validity via a hierarchical tree decomposition.", "tags": [ "drug-discovery", "molecular-generation" ], "tasks": [ "Drug Discovery", "Molecular Generation" ], "modalities": [ "Small Molecule" ], "organism": [], "api": false }, { "id": "matcha", "name": "Matcha", "type": "model", "url": "https://github.com/LigandPro/Matcha", "description": "Multi-stage Riemannian flow matching model for physically valid molecular docking with scoring, pose filtering, and benchmarks.", "tags": [ "drug-discovery", "molecular-generation" ], "tasks": [ "Drug Discovery", "Molecular Generation" ], "modalities": [ "Small Molecule" ], "organism": [], "api": false }, { "id": "mcpinn", "name": "MCPINN", "type": "model", "url": "https://github.com/mhlee0903/multi_channels_PINN", "description": "Drug discovery via compound-protein interaction and machine learning.", "tags": [ "compound-protein-interaction", "drug-discovery" ], "tasks": [ "Compound-Protein Interaction", "Drug Discovery" ], "modalities": [ "Protein", "Small Molecule" ], "organism": [], "api": false }, { "id": "midas", "name": "MIDAS", "type": "model", "url": "https://github.com/labomics/midas", "description": "Mosaic integration and differential accessibility model for single-cell multi-omics that handles arbitrary missing-modality combinations across transcriptomics, chromatin accessibility, and proteomics.", "tags": [ "foundation-models", "multi-omics-foundation-models", "single-cell-foundation-models" ], "tasks": [ "Foundation Model" ], "modalities": [ "Multi-Omics", "Single Cell" ], "organism": [], "api": false }, { "id": "mira", "name": "MIRA", "type": "model", "url": "https://github.com/cistrome/MIRA", "description": "Probabilistic multimodal topic model jointly modeling single-cell transcriptomics and chromatin accessibility for regulatory network inference.", "tags": [ "foundation-models", "multi-omics-foundation-models", "single-cell-foundation-models" ], "tasks": [ "Foundation Model" ], "modalities": [ "Multi-Omics", "Single Cell" ], "organism": [], "api": false }, { "id": "mofa", "name": "MOFA+", "type": "model", "url": "https://github.com/bioFAM/MOFA2", "description": "Multi-Omics Factor Analysis framework identifying shared axes of variation across bulk and single-cell datasets including RNA, ATAC, proteomics, methylation, and copy number.", "tags": [ "foundation-models", "multi-omics-foundation-models", "single-cell-foundation-models" ], "tasks": [ "Foundation Model" ], "modalities": [ "Multi-Omics", "Single Cell" ], "organism": [], "api": false }, { "id": "mofgcn", "name": "MOFGCN", "type": "model", "url": "https://github.com/weiba/MOFGCN/tree/main", "description": "GCN + heterogeneous network.", "tags": [ "drug-discovery", "drug-response-prediction" ], "tasks": [ "Drug Discovery", "Drug Response Prediction" ], "modalities": [ "Small Molecule" ], "organism": [], "api": false }, { "id": "mol2vec", "name": "Mol2Vec", "type": "model", "url": "https://github.com/samoturk/mol2vec", "description": "Unsupervised molecular embedding method inspired by Word2Vec for learning vector representations of chemical substructures.", "tags": [ "compound-embedding", "compound-foundation-models", "foundation-models" ], "tasks": [ "Foundation Model" ], "modalities": [ "Small Molecule" ], "organism": [], "api": false }, { "id": "molecular_transformer", "name": "Molecular Transformer", "type": "model", "url": "https://github.com/pschwllr/MolecularTransformer", "description": "Sequence-to-sequence model for retrosynthesis prediction.", "tags": [ "drug-discovery", "molecular-generation" ], "tasks": [ "Drug Discovery", "Molecular Generation" ], "modalities": [ "Small Molecule" ], "organism": [], "api": false }, { "id": "molformer", "name": "MolFormer", "type": "model", "url": "https://github.com/IBM/molformer", "description": "Linear attention transformer pretrained on millions of SMILES strings for efficient molecular embeddings.", "tags": [ "compound-embedding", "compound-foundation-models", "foundation-models" ], "tasks": [ "representation-learning" ], "modalities": [ "chemical-structure" ], "organism": [], "api": false, "entities": [ "molecule" ], "methods": [ "language-model", "self-supervised-learning", "transformer" ], "github": "https://github.com/IBM/molformer", "last_checked": "2026-08-08", "metadata_sources": [ "https://github.com/IBM/molformer" ] }, { "id": "molgpt", "name": "MolGPT", "type": "model", "url": "https://github.com/devalab/molgpt", "description": "Transformer-based model for molecular generation.", "tags": [ "drug-discovery", "molecular-generation" ], "tasks": [ "Drug Discovery", "Molecular Generation" ], "modalities": [ "Small Molecule" ], "organism": [], "api": false }, { "id": "molt5", "name": "MolT5", "type": "model", "url": "https://github.com/blender-nlp/MolT5", "description": "Language model for molecular tasks bridging text and SMILES, enabling molecule captioning and text-driven molecule generation.", "tags": [ "llm-for-biology" ], "tasks": [ "Language Modeling" ], "modalities": [ "Text" ], "organism": [], "api": false }, { "id": "moltrans", "name": "MolTrans", "type": "model", "url": "https://github.com/kexinhuang12345/MolTrans", "description": "Transformer-based DTI model leveraging molecular substructures.", "tags": [ "drug-discovery", "drug-target-interaction" ], "tasks": [ "Drug Discovery", "Drug Target Interaction" ], "modalities": [ "Protein", "Small Molecule" ], "organism": [], "api": false }, { "id": "multigrate", "name": "Multigrate", "type": "model", "url": "https://github.com/theislab/multigrate", "description": "Asymmetric multi-omics variational autoencoder for integrating single-cell data across RNA, ATAC, and protein modalities with missing-modality support.", "tags": [ "foundation-models", "multi-omics-foundation-models", "single-cell-foundation-models" ], "tasks": [ "Foundation Model" ], "modalities": [ "Multi-Omics", "Single Cell" ], "organism": [], "api": false }, { "id": "multivi", "name": "MultiVI", "type": "model", "url": "https://github.com/scverse/scvi-tools", "description": "Multi-modal variational autoencoder for integrating paired and unpaired single-cell RNA-seq and ATAC-seq measurements into a unified latent space.", "tags": [ "foundation-models", "multi-omics-foundation-models", "single-cell-foundation-models" ], "tasks": [ "Foundation Model" ], "modalities": [ "Multi-Omics", "Single Cell" ], "organism": [], "api": false }, { "id": "musk", "name": "MUSK", "type": "model", "url": "https://github.com/lilab-stanford/MUSK", "description": "Vision-language foundation model for precision oncology analyzing multimodal paired text and pathology image data for biomarker prediction and retrieval.", "tags": [ "foundation-models", "multi-modal-foundation-models" ], "tasks": [ "Foundation Model" ], "modalities": [ "Multi-Modal" ], "organism": [], "api": false }, { "id": "nbbayeslm", "name": "NbBayesLM", "type": "model", "url": "https://github.com/FairuzShadmaniShishir/NbBayesLM", "description": "Bayesian neural network integrating protein language model embeddings and physicochemical features to predict nanobody thermostability with uncertainty estimates. [Paper](https://www.frontiersin.org/journals/bioinformatics/articles/10.3389/fbinf.2026.1832968/full)", "tags": [ "protein-property-prediction" ], "tasks": [ "Protein Property Prediction" ], "modalities": [ "Protein" ], "organism": [], "api": false }, { "id": "neodti", "name": "NeoDTI", "type": "model", "url": "https://github.com/FangpingWan/NeoDTI", "description": "Library for drug-target interaction prediction.", "tags": [ "drug-discovery", "drug-target-interaction" ], "tasks": [ "Drug Discovery", "Drug Target Interaction" ], "modalities": [ "Protein", "Small Molecule" ], "organism": [], "api": false }, { "id": "nicheformer", "name": "Nicheformer", "type": "model", "url": "https://github.com/theislab/nicheformer", "description": "Foundation model for single-cell and spatial omics using a transformer architecture with positional embeddings to encode spatial cell information.", "tags": [ "foundation-models", "single-cell-foundation-models", "spatial-foundation-models" ], "tasks": [ "foundation-model-pretraining", "representation-learning" ], "modalities": [ "single-cell-rna-seq", "spatial-transcriptomics", "transcriptomics" ], "organism": [], "api": false, "entities": [ "cell", "gene", "tissue" ], "methods": [ "self-supervised-learning", "transformer" ], "year": 2024, "github": "https://github.com/theislab/nicheformer", "paper": "https://doi.org/10.1101/2024.04.15.589472", "last_checked": "2026-08-08", "metadata_sources": [ "https://github.com/theislab/nicheformer", "https://doi.org/10.1101/2024.04.15.589472" ] }, { "id": "nucleotide_transformer", "name": "Nucleotide Transformer", "type": "model", "url": "https://github.com/instadeepai/nucleotide-transformer", "description": "Foundation model for genomic sequences across multiple species.", "tags": [ "foundation-models", "genomics-foundation-models" ], "tasks": [ "Foundation Model" ], "modalities": [ "Genomics" ], "organism": [], "api": false }, { "id": "omegafold", "name": "OmegaFold", "type": "model", "url": "https://github.com/HeliXonProtein/OmegaFold", "description": "High-resolution de novo protein structure prediction from sequence.", "tags": [ "foundation-models", "protein-foundation-models", "protein-structure-prediction-and-design" ], "tasks": [ "Foundation Model", "Protein Structure Prediction" ], "modalities": [ "Protein" ], "organism": [], "api": false }, { "id": "openfold", "name": "OpenFold", "type": "model", "url": "https://github.com/aqlaboratory/openfold", "description": "Trainable, memory-efficient open-source reproduction of AlphaFold2 enabling custom protein structure prediction workflows.", "tags": [ "foundation-models", "protein-foundation-models", "protein-structure-prediction-and-design" ], "tasks": [ "Foundation Model", "Protein Structure Prediction" ], "modalities": [ "Protein" ], "organism": [], "api": false }, { "id": "paccmannrl", "name": "PaccMannRL", "type": "model", "url": "https://github.com/PaccMann/paccmann_generator", "description": "Reinforcement learning-based generative model for de novo hit-like anticancer molecule design from transcriptomic data.", "tags": [ "drug-discovery", "molecular-generation" ], "tasks": [ "Drug Discovery", "Molecular Generation" ], "modalities": [ "Small Molecule" ], "organism": [], "api": false }, { "id": "pathomicfusion", "name": "PathomicFusion", "type": "model", "url": "https://github.com/mahmoodlab/PathomicFusion", "description": "Integrated framework fusing histopathology and genomic features via CNN, GNN, and attention gating for cancer diagnosis and prognosis.", "tags": [ "foundation-models", "multi-modal-foundation-models" ], "tasks": [ "Foundation Model" ], "modalities": [ "Multi-Modal" ], "organism": [], "api": false }, { "id": "phikon", "name": "Phikon", "type": "model", "url": "https://huggingface.co/owkin/phikon", "description": "ViT-based pathology foundation model pretrained with iBOT self-supervision on TCGA whole-slide images.", "tags": [ "foundation-models", "single-cell-foundation-models", "spatial-foundation-models" ], "tasks": [ "foundation-model-pretraining", "representation-learning" ], "modalities": [ "histopathology", "imaging" ], "organism": [], "api": false, "entities": [ "tissue" ], "methods": [ "self-supervised-learning", "transformer" ], "documentation": "https://huggingface.co/owkin/phikon", "last_checked": "2026-08-08", "metadata_sources": [ "https://huggingface.co/owkin/phikon" ] }, { "id": "plip", "name": "PLIP", "type": "model", "url": "https://github.com/PathologyFoundation/plip", "description": "Vision-language foundation model for pathology trained with contrastive learning on pathology image–text pairs for image classification and text-to-image retrieval.", "tags": [ "foundation-models", "multi-modal-foundation-models" ], "tasks": [ "classification", "representation-learning" ], "modalities": [ "histopathology", "imaging" ], "organism": [], "api": false, "entities": [ "tissue" ], "methods": [ "contrastive-learning" ], "github": "https://github.com/PathologyFoundation/plip", "last_checked": "2026-08-08", "metadata_sources": [ "https://github.com/PathologyFoundation/plip" ] }, { "id": "porpoise", "name": "PORPOISE", "type": "model", "url": "https://github.com/mahmoodlab/PORPOISE", "description": "Pan-cancer integrative histology-genomic analysis framework using multimodal deep learning for patient stratification.", "tags": [ "foundation-models", "multi-modal-foundation-models" ], "tasks": [ "Foundation Model" ], "modalities": [ "Multi-Modal" ], "organism": [], "api": false }, { "id": "prnet", "name": "PRNet", "type": "model", "url": "https://github.com/Perturbation-Response-Prediction/PRnet", "description": "Deep generative model for predicting transcriptional responses to novel chemical perturbations for drug discovery.", "tags": [ "drug-discovery", "drug-perturbation" ], "tasks": [ "Drug Discovery", "Drug Perturbation" ], "modalities": [ "Small Molecule" ], "organism": [], "api": false }, { "id": "progen2", "name": "ProGen2", "type": "model", "url": "https://github.com/salesforce/progen", "description": "Protein language model trained on diverse protein families for sequence generation and fitness prediction.", "tags": [ "foundation-models", "pre-trained-embedding", "protein-foundation-models" ], "tasks": [ "protein-sequence-design", "representation-learning" ], "modalities": [ "protein-sequence" ], "organism": [], "api": false, "entities": [ "protein" ], "methods": [ "generative-model", "language-model", "transformer" ], "github": "https://github.com/salesforce/progen", "last_checked": "2026-08-08", "metadata_sources": [ "https://github.com/salesforce/progen" ] }, { "id": "proteinmpnn", "name": "ProteinMPNN", "type": "model", "url": "https://github.com/dauparas/ProteinMPNN", "description": "Deep learning model for protein sequence design given backbone structure.", "tags": [ "foundation-models", "protein-foundation-models", "protein-structure-prediction-and-design" ], "tasks": [ "protein-sequence-design" ], "modalities": [ "molecular-structure", "protein-sequence" ], "organism": [], "api": false, "entities": [ "protein" ], "methods": [ "graph-neural-network", "message-passing-neural-network" ], "year": 2022, "github": "https://github.com/dauparas/ProteinMPNN", "last_checked": "2026-08-08", "metadata_sources": [ "https://github.com/dauparas/ProteinMPNN" ] }, { "id": "prottrans", "name": "ProtTrans", "type": "model", "url": "https://github.com/agemagician/ProtTrans", "description": "Suite of protein language models (ProtBERT, ProtT5, ProtXLNet) trained on billions of protein sequences from UniRef and BFD.", "tags": [ "foundation-models", "pre-trained-embedding", "protein-foundation-models" ], "tasks": [ "representation-learning" ], "modalities": [ "protein-sequence" ], "organism": [], "api": false, "entities": [ "protein" ], "methods": [ "language-model", "self-supervised-learning", "transformer" ], "github": "https://github.com/agemagician/ProtTrans", "last_checked": "2026-08-08", "metadata_sources": [ "https://github.com/agemagician/ProtTrans" ] }, { "id": "recover", "name": "RECOVER", "type": "model", "url": "https://github.com/RECOVERcoalition/Recover", "description": "Machine learning framework for predicting synergistic drug combination responses across cell lines.", "tags": [ "drug-discovery", "drug-response-prediction" ], "tasks": [ "Drug Discovery", "Drug Response Prediction" ], "modalities": [ "Small Molecule" ], "organism": [], "api": false }, { "id": "reinvent", "name": "REINVENT", "type": "model", "url": "https://github.com/MolecularAI/Reinvent", "description": "Reinforcement learning for de novo drug design.", "tags": [ "drug-discovery", "molecular-generation" ], "tasks": [ "Drug Discovery", "Molecular Generation" ], "modalities": [ "Small Molecule" ], "organism": [], "api": false }, { "id": "release", "name": "ReLeaSE", "type": "model", "url": "https://github.com/isayev/ReLeaSE", "description": "Deep reinforcement learning framework for de novo drug design combining a generative and predictive model.", "tags": [ "drug-discovery", "molecular-generation" ], "tasks": [ "Drug Discovery", "Molecular Generation" ], "modalities": [ "Small Molecule" ], "organism": [], "api": false }, { "id": "rfdiffusion", "name": "RFdiffusion", "type": "model", "url": "https://github.com/RosettaCommons/RFdiffusion", "description": "Generative model for protein backbone design using diffusion.", "tags": [ "foundation-models", "protein-foundation-models", "protein-structure-prediction-and-design" ], "tasks": [ "Foundation Model", "Protein Structure Prediction" ], "modalities": [ "Protein" ], "organism": [], "api": false }, { "id": "rosettafold", "name": "RoseTTAFold", "type": "model", "url": "https://github.com/RosettaCommons/RoseTTAFold", "description": "Three-track neural network for protein structure prediction.", "tags": [ "foundation-models", "protein-foundation-models", "protein-structure-prediction-and-design" ], "tasks": [ "Foundation Model", "Protein Structure Prediction" ], "modalities": [ "Protein" ], "organism": [], "api": false }, { "id": "saprot", "name": "SaProt", "type": "model", "url": "https://github.com/westlake-reup/SaProt", "description": "Structure-aware protein language model using structure-aware tokens that encode both sequence and backbone geometry for improved function prediction.", "tags": [ "foundation-models", "protein-foundation-models", "protein-structure-prediction-and-design" ], "tasks": [ "Foundation Model", "Protein Structure Prediction" ], "modalities": [ "Protein" ], "organism": [], "api": false }, { "id": "saturn", "name": "SATURN", "type": "model", "url": "https://github.com/snap-stanford/SATURN", "description": "Transformer-based model integrating gene expression and protein sequences via a protein language model to learn unified multi-species cell embeddings.", "tags": [ "foundation-models", "single-cell-foundation-models", "transcriptomics-foundation-models" ], "tasks": [ "Foundation Model" ], "modalities": [ "Single Cell", "Transcriptomics" ], "organism": [], "api": false }, { "id": "scarches", "name": "scArches", "type": "model", "url": "https://github.com/theislab/scarches", "description": "Transfer learning framework for mapping new single-cell datasets onto pre-trained reference atlases across batches, conditions, and modalities.", "tags": [ "domain-alignment", "foundation-models", "single-cell-foundation-models" ], "tasks": [ "Domain Alignment", "Foundation Model" ], "modalities": [ "Single Cell" ], "organism": [], "api": false }, { "id": "scbert", "name": "scBERT", "type": "model", "url": "https://github.com/TencentAILabHealthcare/scBERT", "description": "BERT-based foundation model pretrained on large-scale scRNA-seq data for cell type annotation.", "tags": [ "foundation-models", "single-cell-foundation-models", "transcriptomics-foundation-models" ], "tasks": [ "cell-type-annotation", "classification", "foundation-model-pretraining" ], "modalities": [ "single-cell-rna-seq", "transcriptomics" ], "organism": [], "api": false, "entities": [ "cell", "gene" ], "methods": [ "language-model", "self-supervised-learning", "transformer" ], "year": 2022, "github": "https://github.com/TencentAILabHealthcare/scBERT", "paper": "https://www.nature.com/articles/s42256-022-00534-z", "last_checked": "2026-08-08", "metadata_sources": [ "https://github.com/TencentAILabHealthcare/scBERT", "https://www.nature.com/articles/s42256-022-00534-z" ] }, { "id": "scbutterfly", "name": "scButterfly", "type": "model", "url": "https://github.com/BioX-NKU/scButterfly", "description": "Dual-aligned variational autoencoder for single-cell cross-modality translation between paired and unpaired multiomics data.", "tags": [ "foundation-models", "multi-omics-foundation-models", "single-cell-foundation-models" ], "tasks": [ "Foundation Model" ], "modalities": [ "Multi-Omics", "Single Cell" ], "organism": [], "api": false }, { "id": "scfoundation", "name": "scFoundation", "type": "model", "url": "https://github.com/biomap-research/scFoundation", "description": "Large-scale foundation model for single-cell gene expression, enabling multiple downstream tasks.", "tags": [ "foundation-models", "single-cell-foundation-models", "transcriptomics-foundation-models" ], "tasks": [ "cell-type-annotation", "drug-response-prediction", "foundation-model-pretraining", "perturbation-prediction", "representation-learning" ], "modalities": [ "single-cell-rna-seq", "transcriptomics" ], "organism": [], "api": false, "entities": [ "cell", "gene" ], "methods": [ "self-supervised-learning", "transformer" ], "year": 2024, "github": "https://github.com/biomap-research/scFoundation", "paper": "https://www.nature.com/articles/s41592-024-02305-7", "last_checked": "2026-08-08", "metadata_sources": [ "https://github.com/biomap-research/scFoundation", "https://www.nature.com/articles/s41592-024-02305-7" ] }, { "id": "scgpt", "name": "scGPT", "type": "model", "url": "https://github.com/bowang-lab/scGPT", "description": "Transformer-based foundation model pretrained on millions of single-cell profiles.", "tags": [ "foundation-models", "single-cell-foundation-models", "transcriptomics-foundation-models" ], "tasks": [ "cell-type-annotation", "foundation-model-pretraining", "gene-regulatory-network-inference", "perturbation-prediction", "representation-learning" ], "modalities": [ "multi-omics", "single-cell-rna-seq", "transcriptomics" ], "organism": [], "api": false, "entities": [ "cell", "gene" ], "methods": [ "generative-model", "self-supervised-learning", "transformer" ], "year": 2024, "github": "https://github.com/bowang-lab/scGPT", "documentation": "https://scgpt.readthedocs.io/en/latest/", "paper": "https://www.nature.com/articles/s41592-024-02201-0", "last_checked": "2026-08-08", "metadata_sources": [ "https://github.com/bowang-lab/scGPT", "https://www.nature.com/articles/s41592-024-02201-0" ] }, { "id": "scgpt_spatial", "name": "scGPT-spatial", "type": "model", "url": "https://github.com/bowang-lab/scGPT-spatial", "description": "Extension of scGPT for spatial transcriptomics with continual pretraining and a mixture-of-experts decoder for spatial gene expression analysis.", "tags": [ "foundation-models", "single-cell-foundation-models", "spatial-foundation-models" ], "tasks": [ "foundation-model-pretraining", "imputation", "representation-learning" ], "modalities": [ "multi-omics", "single-cell-rna-seq", "spatial-transcriptomics" ], "organism": [], "api": false, "entities": [ "cell", "gene", "tissue" ], "methods": [ "generative-model", "self-supervised-learning", "transformer" ], "year": 2025, "github": "https://github.com/bowang-lab/scGPT-spatial", "paper": "https://www.biorxiv.org/content/10.1101/2025.02.05.636714v1", "last_checked": "2026-08-08", "metadata_sources": [ "https://github.com/bowang-lab/scGPT-spatial", "https://www.biorxiv.org/content/10.1101/2025.02.05.636714v1" ] }, { "id": "scmulan", "name": "scMulan", "type": "model", "url": "https://github.com/SuperBianC/scMulan", "description": "Single-cell multi-omic language model pretrained on ~10M cells spanning transcriptomics, epigenomics, and proteomics for cross-omics transfer tasks.", "tags": [ "foundation-models", "multi-omics-foundation-models", "single-cell-foundation-models" ], "tasks": [ "foundation-model-pretraining", "representation-learning" ], "modalities": [ "epigenomics", "multi-omics", "proteomics", "single-cell-rna-seq", "transcriptomics" ], "organism": [], "api": false, "entities": [ "cell", "gene" ], "methods": [ "language-model", "transformer" ], "github": "https://github.com/SuperBianC/scMulan", "last_checked": "2026-08-08", "metadata_sources": [ "https://github.com/SuperBianC/scMulan" ] }, { "id": "scpair", "name": "scPair", "type": "model", "url": "https://github.com/quon-titative-biology/scPair", "description": "Bidirectional feedforward network for single-cell multimodal analysis with cross-modality prediction leveraging single-cell atlases.", "tags": [ "foundation-models", "multi-omics-foundation-models", "single-cell-foundation-models" ], "tasks": [ "Foundation Model" ], "modalities": [ "Multi-Omics", "Single Cell" ], "organism": [], "api": false }, { "id": "scprint", "name": "scPRINT", "type": "model", "url": "https://github.com/cantinilab/scPRINT", "description": "Pretrained on 50M cells for scRNA-seq denoising & zero imputation.", "tags": [ "llm-for-biology" ], "tasks": [ "batch-correction", "cell-type-annotation", "foundation-model-pretraining", "gene-regulatory-network-inference", "imputation", "representation-learning" ], "modalities": [ "single-cell-rna-seq", "transcriptomics" ], "organism": [], "api": false, "entities": [ "cell", "gene" ], "methods": [ "self-supervised-learning", "transformer" ], "year": 2025, "github": "https://github.com/cantinilab/scPRINT", "documentation": "https://www.jkobject.com/scPRINT/", "paper": "https://www.nature.com/articles/s41467-025-58699-1", "last_checked": "2026-08-08", "metadata_sources": [ "https://github.com/cantinilab/scPRINT", "https://www.nature.com/articles/s41467-025-58699-1" ] }, { "id": "sei", "name": "Sei", "type": "model", "url": "https://github.com/FunctionLab/sei-framework", "description": "Sequence-to-function framework learning a genome-wide regulatory activity code from DNA sequences for variant effect prediction.", "tags": [ "foundation-models", "genomics-foundation-models" ], "tasks": [ "Foundation Model" ], "modalities": [ "Genomics" ], "organism": [], "api": false }, { "id": "spatialglue", "name": "SpatialGlue", "type": "model", "url": "https://github.com/zhanglabtools/SpatialGlue", "description": "Graph attention network for spatial multi-omics integration jointly embedding spatial transcriptomics with chromatin accessibility or proteomics.", "tags": [ "foundation-models", "multi-omics-foundation-models", "single-cell-foundation-models" ], "tasks": [ "Foundation Model" ], "modalities": [ "Multi-Omics", "Single Cell" ], "organism": [], "api": false }, { "id": "targetdiff", "name": "TargetDiff", "type": "model", "url": "https://github.com/guanjq/targetdiff", "description": "3D equivariant diffusion model for structure-based drug design.", "tags": [ "drug-discovery", "molecular-generation" ], "tasks": [ "Drug Discovery", "Molecular Generation" ], "modalities": [ "Small Molecule" ], "organism": [], "api": false }, { "id": "tgsa", "name": "TGSA", "type": "model", "url": "https://github.com/violet-sto/TGSA", "description": "Tumor gene set and attention-based model leveraging biological pathway knowledge for drug response prediction.", "tags": [ "drug-discovery", "drug-response-prediction" ], "tasks": [ "Drug Discovery", "Drug Response Prediction" ], "modalities": [ "Small Molecule" ], "organism": [], "api": false }, { "id": "toad", "name": "TOAD", "type": "model", "url": "https://github.com/mahmoodlab/TOAD", "description": "Tumor Origin Assessment via Deep-learning; weakly-supervised multi-task model predicting cancer primary origin from H&E whole-slide images.", "tags": [ "foundation-models", "multi-modal-foundation-models" ], "tasks": [ "Foundation Model" ], "modalities": [ "Multi-Modal" ], "organism": [], "api": false }, { "id": "tosica", "name": "TOSICA", "type": "model", "url": "https://github.com/JackieHanlaopo/TOSICA", "description": "Transformer-based framework for one-stop interpretable cell-type annotation supporting cross-dataset and cross-species transfer.", "tags": [ "domain-alignment", "foundation-models", "single-cell-foundation-models" ], "tasks": [ "Domain Alignment", "Foundation Model" ], "modalities": [ "Single Cell" ], "organism": [], "api": false }, { "id": "totalvi", "name": "totalVI", "type": "model", "url": "https://github.com/scverse/scvi-tools", "description": "Probabilistic framework for joint analysis of paired scRNA-seq and protein (CITE-seq) data enabling multi-modal cell state representation across single-cell datasets.", "tags": [ "foundation-models", "multi-omics-foundation-models", "single-cell-foundation-models" ], "tasks": [ "Foundation Model" ], "modalities": [ "Multi-Omics", "Single Cell" ], "organism": [], "api": false }, { "id": "transformercpi", "name": "TransformerCPI", "type": "model", "url": "https://github.com/lifanchen-simm/transformerCPI", "description": "CPI prediction using Transformer.", "tags": [ "compound-protein-interaction", "drug-discovery" ], "tasks": [ "Compound-Protein Interaction", "Drug Discovery" ], "modalities": [ "Protein", "Small Molecule" ], "organism": [], "api": false }, { "id": "transigen", "name": "TranSiGen", "type": "model", "url": "https://github.com/myzhengSIMM/TranSiGen", "description": "Dual-VAE architecture for ligand-based virtual screening, drug response prediction, and drug repurposing using chemical-induced transcriptional profiles.", "tags": [ "drug-discovery", "drug-repurposing" ], "tasks": [ "Drug Discovery", "Drug Repurposing" ], "modalities": [ "Small Molecule" ], "organism": [], "api": false }, { "id": "uce", "name": "UCE", "type": "model", "url": "https://github.com/snap-stanford/UCE", "description": "Universal Cell Embeddings: zero-shot single-cell embedding model trained on 36M cells across species, tissues, and assays without fine-tuning.", "tags": [ "foundation-models", "single-cell-foundation-models", "transcriptomics-foundation-models" ], "tasks": [ "foundation-model-pretraining", "representation-learning" ], "modalities": [ "single-cell-rna-seq", "transcriptomics" ], "organism": [], "api": false, "entities": [ "cell" ], "methods": [ "self-supervised-learning" ], "year": 2026, "github": "https://github.com/snap-stanford/UCE", "paper": "https://www.nature.com/articles/s41586-026-10689-z", "last_checked": "2026-08-08", "metadata_sources": [ "https://github.com/snap-stanford/UCE", "https://www.nature.com/articles/s41586-026-10689-z" ] }, { "id": "uni", "name": "UNI", "type": "model", "url": "https://github.com/mahmoodlab/UNI", "description": "General-purpose self-supervised pathology foundation model trained on 100K+ whole-slide images for diverse computational pathology tasks.", "tags": [ "foundation-models", "single-cell-foundation-models", "spatial-foundation-models" ], "tasks": [ "foundation-model-pretraining", "representation-learning" ], "modalities": [ "histopathology", "imaging" ], "organism": [], "api": false, "entities": [ "tissue" ], "methods": [ "self-supervised-learning", "transformer" ], "github": "https://github.com/mahmoodlab/UNI", "last_checked": "2026-08-08", "metadata_sources": [ "https://github.com/mahmoodlab/UNI" ] }, { "id": "uni_mol", "name": "Uni-Mol", "type": "model", "url": "https://github.com/deepmodeling/Uni-Mol", "description": "3D molecular pretraining framework for universal representation learning on molecules and protein pockets.", "tags": [ "compound-embedding", "compound-foundation-models", "foundation-models" ], "tasks": [ "docking", "representation-learning" ], "modalities": [ "chemical-structure", "molecular-structure" ], "organism": [], "api": false, "entities": [ "molecule", "protein" ], "methods": [ "self-supervised-learning", "transformer" ], "year": 2023, "github": "https://github.com/deepmodeling/Uni-Mol", "paper": "https://openreview.net/forum?id=6K2RM6wVqKu", "last_checked": "2026-08-08", "metadata_sources": [ "https://github.com/deepmodeling/Uni-Mol", "https://openreview.net/forum?id=6K2RM6wVqKu" ] }, { "id": "unitednet", "name": "UnitedNet", "type": "model", "url": "https://github.com/LiuLab-Bioelectronics-Harvard/UnitedNet", "description": "Interpretable multi-task deep neural network for single-cell multi-omics integration spanning transcriptomics, chromatin accessibility, and proteomics.", "tags": [ "foundation-models", "multi-omics-foundation-models", "single-cell-foundation-models" ], "tasks": [ "Foundation Model" ], "modalities": [ "Multi-Omics", "Single Cell" ], "organism": [], "api": false }, { "id": "virchow", "name": "Virchow", "type": "model", "url": "https://huggingface.co/paige-ai/Virchow", "description": "Million-slide digital pathology foundation model using a vision transformer and self-supervised distillation for tile-level pathology image representation.", "tags": [ "foundation-models", "multi-modal-foundation-models" ], "tasks": [ "Foundation Model" ], "modalities": [ "Multi-Modal" ], "organism": [], "api": false }, { "id": "autozyme", "name": "AutoZyme", "type": "toolkit", "url": "https://github.com/ElliotXie/autozyme", "description": "Autonomous agentic framework that speeds up bioinformatics software (e.g. Scanpy, Seurat) on CPUs while preserving the original results.", "tags": [ "preprocessing-tools" ], "tasks": [ "Preprocessing" ], "modalities": [], "organism": [], "api": false }, { "id": "biopython", "name": "Biopython", "type": "toolkit", "url": "https://biopython.org/", "description": "Collection of Python tools for biological computation including sequence analysis, structure parsing, and database access.", "tags": [ "preprocessing-tools" ], "tasks": [ "Preprocessing" ], "modalities": [ "dna-sequence", "protein-sequence" ], "organism": [], "api": false, "entities": [ "gene", "protein" ], "github": "https://github.com/biopython/biopython", "documentation": "https://biopython.org/wiki/Documentation", "last_checked": "2026-08-08", "metadata_sources": [ "https://github.com/biopython/biopython", "https://biopython.org/" ] }, { "id": "casper", "name": "CaSpER", "type": "toolkit", "url": "https://github.com/akdess/CaSpER", "description": "CNV identification and visualization by integrative analysis of single-cell or bulk RNA-seq data.", "tags": [ "preprocessing-tools" ], "tasks": [ "Preprocessing" ], "modalities": [], "organism": [], "api": false }, { "id": "cellcharter", "name": "CellCharter", "type": "toolkit", "url": "https://github.com/CSOgroup/cellcharter", "description": "Identification and characterization of spatial cell niches from spatial transcriptomics using VAEs and Gaussian mixture models.", "tags": [ "preprocessing-tools" ], "tasks": [ "Preprocessing" ], "modalities": [ "spatial-transcriptomics" ], "organism": [], "api": false, "entities": [ "cell", "tissue" ], "github": "https://github.com/CSOgroup/cellcharter", "documentation": "https://cellcharter.readthedocs.io/", "last_checked": "2026-08-08", "metadata_sources": [ "https://github.com/CSOgroup/cellcharter", "https://cellcharter.readthedocs.io/" ] }, { "id": "cellchat", "name": "CellChat", "type": "toolkit", "url": "https://github.com/sqjin/CellChat", "description": "Inference and analysis of cell-cell communication ligand-receptor networks from single-cell transcriptomics data.", "tags": [ "preprocessing-tools" ], "tasks": [ "Preprocessing" ], "modalities": [ "single-cell-rna-seq" ], "organism": [], "api": false, "entities": [ "cell", "gene" ], "github": "https://github.com/sqjin/CellChat", "last_checked": "2026-08-08", "metadata_sources": [ "https://github.com/sqjin/CellChat" ] }, { "id": "celltypist", "name": "CellTypist", "type": "toolkit", "url": "https://github.com/Teichlab/celltypist", "description": "Automated cell type annotation for scRNA-seq.", "tags": [ "preprocessing-tools" ], "tasks": [ "cell-type-annotation" ], "modalities": [ "single-cell-rna-seq" ], "organism": [], "api": false, "entities": [ "cell", "gene" ], "github": "https://github.com/Teichlab/celltypist", "documentation": "https://celltypist.readthedocs.io/", "last_checked": "2026-08-08", "metadata_sources": [ "https://github.com/Teichlab/celltypist", "https://celltypist.readthedocs.io/" ] }, { "id": "chatspatial", "name": "ChatSpatial", "type": "toolkit", "url": "https://github.com/cafferychen777/ChatSpatial", "description": "MCP server for spatial transcriptomics analysis via natural language.", "tags": [ "preprocessing-tools" ], "tasks": [ "Preprocessing" ], "modalities": [], "organism": [], "api": false }, { "id": "chemistry_development_kit", "name": "Chemistry Development Kit", "type": "toolkit", "url": "https://github.com/cdk/cdk", "description": "Cheminformatics software & machine learning tools.", "tags": [ "preprocessing-tools" ], "tasks": [ "Preprocessing" ], "modalities": [], "organism": [], "api": false }, { "id": "commot", "name": "COMMOT", "type": "toolkit", "url": "https://github.com/zcang/COMMOT", "description": "Optimal transport-based framework for screening cell-cell communication in spatial transcriptomics.", "tags": [ "preprocessing-tools" ], "tasks": [ "Preprocessing" ], "modalities": [], "organism": [], "api": false }, { "id": "deepchem", "name": "DeepChem", "type": "toolkit", "url": "https://github.com/deepchem/deepchem", "description": "Deep learning library for drug discovery, quantum chemistry, and materials science.", "tags": [ "preprocessing-tools" ], "tasks": [ "Preprocessing" ], "modalities": [ "chemical-structure", "molecular-structure" ], "organism": [], "api": false, "entities": [ "molecule", "protein" ], "github": "https://github.com/deepchem/deepchem", "documentation": "https://deepchem.readthedocs.io/", "last_checked": "2026-08-08", "metadata_sources": [ "https://github.com/deepchem/deepchem", "https://deepchem.readthedocs.io/" ] }, { "id": "deeptalk", "name": "DeepTalk", "type": "toolkit", "url": "https://github.com/JiangBioLab/DeepTalk", "description": "Graph attention network for deciphering cell-cell communication from spatial transcriptomics.", "tags": [ "preprocessing-tools" ], "tasks": [ "Preprocessing" ], "modalities": [], "organism": [], "api": false }, { "id": "doubletfinder", "name": "DoubletFinder", "type": "toolkit", "url": "https://github.com/chris-mcginnis-ucsf/DoubletFinder", "description": "Machine learning approach for detecting multiplet (doublet) artifacts in single-cell RNA-seq data.", "tags": [ "preprocessing-tools" ], "tasks": [ "Preprocessing" ], "modalities": [], "organism": [], "api": false }, { "id": "flashdeconv", "name": "FlashDeconv", "type": "toolkit", "url": "https://github.com/cafferychen777/flashdeconv", "description": "High-performance spatial transcriptomics deconvolution (~1M spots in ~3 min).", "tags": [ "preprocessing-tools" ], "tasks": [ "Preprocessing" ], "modalities": [], "organism": [], "api": false }, { "id": "gromacs", "name": "GROMACS", "type": "toolkit", "url": "https://www.gromacs.org/", "description": "Molecular dynamics simulation package for biochemical molecules.", "tags": [ "preprocessing-tools" ], "tasks": [ "Preprocessing" ], "modalities": [], "organism": [], "api": false }, { "id": "harmony", "name": "Harmony", "type": "toolkit", "url": "https://github.com/immunogenomics/harmony", "description": "Fast and scalable integration of single-cell data across datasets, conditions, technologies, and species.", "tags": [ "preprocessing-tools" ], "tasks": [ "Preprocessing" ], "modalities": [], "organism": [], "api": false }, { "id": "kallisto", "name": "kallisto", "type": "toolkit", "url": "https://pachterlab.github.io/kallisto/", "description": "Near-optimal RNA-seq quantification using pseudoalignment for fast transcript abundance estimation.", "tags": [ "preprocessing-tools" ], "tasks": [ "Preprocessing" ], "modalities": [], "organism": [], "api": false }, { "id": "linger", "name": "LINGER", "type": "toolkit", "url": "https://github.com/Durenlab/LINGER", "description": "Neural network for gene regulatory network inference from single-cell multiome (RNA+ATAC-seq) data with bulk data pretraining.", "tags": [ "preprocessing-tools" ], "tasks": [ "Preprocessing" ], "modalities": [], "organism": [], "api": false }, { "id": "mdanalysis", "name": "MDAnalysis", "type": "toolkit", "url": "https://www.mdanalysis.org/", "description": "Python library for analyzing and altering molecular dynamics simulation trajectories.", "tags": [ "preprocessing-tools" ], "tasks": [ "Preprocessing" ], "modalities": [], "organism": [], "api": false }, { "id": "mogonet", "name": "MOGONET", "type": "toolkit", "url": "https://github.com/txWang/MOGONET", "description": "Multi-omics graph convolutional network framework for patient classification and biomarker identification.", "tags": [ "preprocessing-tools" ], "tasks": [ "Preprocessing" ], "modalities": [], "organism": [], "api": false }, { "id": "monocle3", "name": "Monocle3", "type": "toolkit", "url": "https://cole-trapnell-lab.github.io/monocle3/", "description": "Single-cell trajectory analysis tool for learning developmental trajectories and ordering cells in pseudotime.", "tags": [ "preprocessing-tools" ], "tasks": [ "Preprocessing" ], "modalities": [], "organism": [], "api": false }, { "id": "ncem", "name": "NCEM", "type": "toolkit", "url": "https://github.com/theislab/ncem", "description": "GNN-based model for learning intercellular communication from spatial graphs of cells.", "tags": [ "preprocessing-tools" ], "tasks": [ "Preprocessing" ], "modalities": [], "organism": [], "api": false }, { "id": "numbat", "name": "Numbat", "type": "toolkit", "url": "https://github.com/kharchenkolab/numbat", "description": "Haplotype-aware copy number variation inference from single-cell RNA-seq using hidden Markov models.", "tags": [ "preprocessing-tools" ], "tasks": [ "Preprocessing" ], "modalities": [], "organism": [], "api": false }, { "id": "openmm", "name": "OpenMM", "type": "toolkit", "url": "https://openmm.org/", "description": "High-performance toolkit for molecular simulation and GPU-accelerated MD.", "tags": [ "preprocessing-tools" ], "tasks": [ "Preprocessing" ], "modalities": [], "organism": [], "api": false }, { "id": "rdkit", "name": "RDKit", "type": "toolkit", "url": "https://github.com/rdkit/rdkit", "description": "Cheminformatics software & machine learning toolkit.", "tags": [ "preprocessing-tools" ], "tasks": [ "Preprocessing" ], "modalities": [ "chemical-structure" ], "organism": [], "api": false, "entities": [ "molecule" ], "github": "https://github.com/rdkit/rdkit", "documentation": "https://www.rdkit.org/docs/", "last_checked": "2026-08-08", "metadata_sources": [ "https://github.com/rdkit/rdkit", "https://www.rdkit.org/docs/" ] }, { "id": "scanpy", "name": "Scanpy", "type": "toolkit", "url": "https://scanpy.readthedocs.io/en/stable/", "description": "Python library for scRNA-seq analysis.", "tags": [ "preprocessing-tools" ], "tasks": [ "Preprocessing" ], "modalities": [ "single-cell-rna-seq" ], "organism": [], "api": false, "entities": [ "cell", "gene" ], "github": "https://github.com/scverse/scanpy", "documentation": "https://scanpy.readthedocs.io/en/stable/", "last_checked": "2026-08-08", "metadata_sources": [ "https://github.com/scverse/scanpy", "https://scanpy.readthedocs.io/en/stable/" ] }, { "id": "scenic", "name": "SCENIC", "type": "toolkit", "url": "https://github.com/aertslab/SCENIC", "description": "Single-cell regulatory network inference and clustering linking transcription factors to co-expressed gene modules.", "tags": [ "preprocessing-tools" ], "tasks": [ "Preprocessing" ], "modalities": [], "organism": [], "api": false }, { "id": "scipenn", "name": "sciPENN", "type": "toolkit", "url": "https://github.com/jlakkis/sciPENN", "description": "RNN-based method for simultaneous protein expression prediction, uncertainty estimation, and cell-type label transfer from CITE-seq and scRNA-seq data.", "tags": [ "preprocessing-tools" ], "tasks": [ "Preprocessing" ], "modalities": [], "organism": [], "api": false }, { "id": "scvelo", "name": "scVelo", "type": "toolkit", "url": "https://github.com/theislab/scvelo", "description": "RNA velocity estimation for single-cell transcriptomics, inferring the direction and speed of cell differentiation.", "tags": [ "preprocessing-tools" ], "tasks": [ "Preprocessing" ], "modalities": [], "organism": [], "api": false }, { "id": "scvi_tools", "name": "scvi-tools", "type": "toolkit", "url": "https://scvi-tools.org/", "description": "Probabilistic models for single-cell omics data analysis.", "tags": [ "preprocessing-tools" ], "tasks": [ "Preprocessing" ], "modalities": [ "single-cell-rna-seq", "multi-omics" ], "organism": [], "api": false, "entities": [ "cell", "gene" ], "github": "https://github.com/scverse/scvi-tools", "documentation": "https://docs.scvi-tools.org/", "last_checked": "2026-08-08", "metadata_sources": [ "https://github.com/scverse/scvi-tools", "https://docs.scvi-tools.org/" ] }, { "id": "seqbench", "name": "SeqBench", "type": "toolkit", "url": "https://seqbench.com/", "description": "Web-based molecular biology sequence workbench for primer design, cloning simulation (Gibson, Golden Gate, restriction digest), CRISPR guide RNA design, and sequence analysis, with a public REST API, OpenAPI 3.1 spec, and MCP server.", "tags": [ "preprocessing-tools" ], "tasks": [ "Preprocessing" ], "modalities": [], "organism": [], "api": false }, { "id": "seurat", "name": "Seurat", "type": "toolkit", "url": "https://satijalab.org/seurat/", "description": "R library for scRNA-seq analysis.", "tags": [ "preprocessing-tools" ], "tasks": [ "Preprocessing" ], "modalities": [ "single-cell-rna-seq" ], "organism": [], "api": false, "entities": [ "cell", "gene" ], "github": "https://github.com/satijalab/seurat", "documentation": "https://satijalab.org/seurat/", "last_checked": "2026-08-08", "metadata_sources": [ "https://github.com/satijalab/seurat", "https://satijalab.org/seurat/" ] }, { "id": "squidpy", "name": "Squidpy", "type": "toolkit", "url": "https://squidpy.readthedocs.io/", "description": "Python library for spatial single-cell analysis.", "tags": [ "preprocessing-tools" ], "tasks": [ "Preprocessing" ], "modalities": [ "spatial-transcriptomics" ], "organism": [], "api": false, "entities": [ "cell", "tissue" ], "github": "https://github.com/scverse/squidpy", "documentation": "https://squidpy.readthedocs.io/", "last_checked": "2026-08-08", "metadata_sources": [ "https://github.com/scverse/squidpy", "https://squidpy.readthedocs.io/" ] }, { "id": "stagate", "name": "STAGATE", "type": "toolkit", "url": "https://github.com/RucDongLab/STAGATE", "description": "Adaptive graph attention auto-encoder for spatial domain identification in spatial transcriptomics.", "tags": [ "preprocessing-tools" ], "tasks": [ "Preprocessing" ], "modalities": [], "organism": [], "api": false }, { "id": "star", "name": "STAR", "type": "toolkit", "url": "https://github.com/alexdobin/STAR", "description": "Ultrafast universal RNA-seq aligner with support for spliced alignment and single-cell quantification via STARsolo.", "tags": [ "preprocessing-tools" ], "tasks": [ "Preprocessing" ], "modalities": [], "organism": [], "api": false }, { "id": "tigon", "name": "TIGON", "type": "toolkit", "url": "https://github.com/yutongo/TIGON", "description": "Neural optimal transport method for reconstructing growth and dynamic trajectories from single-cell transcriptomics.", "tags": [ "preprocessing-tools" ], "tasks": [ "Preprocessing" ], "modalities": [], "organism": [], "api": false } ]