Files

3275 lines
105 KiB
YAML
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
---
title: "Awesome Computational Biology - machine-readable resource list"
task: ""
lineage_type: import
upstream_source: https://github.com/inoue0426/awesome-computational-biology/blob/c6f07d90/data/resources.yml
upstream_sha: c6f07d90
imported_at: 2026-08-31
prompt_class: catalogue
upstream_changes: accepted
author: upstream
validated: false
---
# Awesome Computational Biology - machine-readable resource list
# Fields
# id : unique slug (required)
# name : display name (required)
# type : category, e.g. database | tool | model | benchmark | api (required)
# url : canonical URL (required)
# description : one-line description (required)
# license : SPDX identifier or free-text (optional)
# api : true | false - whether a programmatic API is available (default: false)
# updated : last-known update date as string YYYY-MM-DD (optional)
# tasks : list of ML/bio tasks (optional)
# modalities : list of data modalities (optional)
# tags : additional free-form tags (optional)
# organism : list of organisms covered (optional)
# paper : DOI or URL to primary publication (optional)
resources:
- id: chembl_web_services
name: "ChEMBL Web Services"
type: api
url: https://www.ebi.ac.uk/chembl/ws
description: "REST API for bioactive molecules, targets, and bioassays."
tags: [api]
tasks: []
modalities: []
organism: []
api: true
- id: clinicaltrials_gov_api
name: "ClinicalTrials.gov API"
type: api
url: https://clinicaltrials.gov/api/gui
description: "API for querying clinical trial metadata and results."
tags: [api]
tasks: []
modalities: []
organism: []
api: true
- id: ensembl_rest_api
name: "Ensembl REST API"
type: api
url: https://rest.ensembl.org/
description: "API for genomic annotations, variants, genes, and comparative genomics."
tags: [api]
tasks: []
modalities: []
organism: []
api: true
- id: kegg_rest_api
name: "KEGG REST API"
type: api
url: https://www.kegg.jp/kegg/rest/keggapi.html
description: "API for accessing KEGG pathways, compounds, genes, and reactions."
tags: [api]
tasks: []
modalities: []
organism: []
api: true
- id: ncbi_e_utilities
name: "NCBI E-utilities"
type: api
url: https://www.ncbi.nlm.nih.gov/books/NBK25501/
description: "Unified APIs for accessing NCBI databases (Gene, GEO, SRA, PubChem, etc)."
tags: [api]
tasks: []
modalities: []
organism: []
api: true
- id: open_targets_platform_api
name: "Open Targets Platform API"
type: api
url: https://platform.opentargets.org/api
description: "API for targetโ€“disease associations integrating genetics, genomics, and drug data."
tags: [api]
tasks: []
modalities: []
organism: []
api: true
- id: pubmed_e_utilities_esearch_efetch
name: "PubMed E-utilities (esearch/efetch)"
type: api
url: https://www.nlm.nih.gov/dataguide/edirect/esearch.html
description: "APIs for searching and retrieving biomedical literature from PubMed."
tags: [api]
tasks: []
modalities: []
organism: []
api: true
- id: uniprot_rest_api
name: "UniProt REST API"
type: api
url: https://www.uniprot.org/help/api
description: "Programmatic access to protein sequence and functional annotation data."
tags: [api]
tasks: []
modalities: []
organism: []
api: true
- id: 1000_genomes_project
name: "1000 Genomes Project"
type: benchmark
url: https://www.internationalgenome.org/
description: "Reference panel of human genetic variation from 2,504 individuals across 26 populations."
tags: [benchmarks-and-datasets]
tasks: []
modalities: []
organism: []
api: false
- id: bace
name: "BACE"
type: benchmark
url: https://www.kaggle.com/datasets/gokturkkoch/bace
description: "Binary classification and regression dataset for ฮฒ-secretase 1 (BACE-1) inhibitor binding affinity."
tags: [benchmarks-and-datasets]
tasks: []
modalities: []
organism: []
api: false
- id: beat_aml
name: "BEAT AML"
type: benchmark
url: https://biodev.github.io/BeatAML2/
description: "Functional ex vivo drug sensitivity measurements paired with genomics for acute myeloid leukemia."
tags: [benchmarks-and-datasets]
tasks: []
modalities: []
organism: []
api: false
- id: bento
name: "Bento"
type: benchmark
url: https://github.com/LigandPro/Bento
description: "Protein-ligand docking benchmark covering rigid, flexible, de novo, blind, induced-fit, and covalent docking tasks."
tags: [benchmarks-and-datasets]
tasks: []
modalities: []
organism: []
api: false
- id: bindingdb_curated_sets
name: "BindingDB Curated Sets"
type: benchmark
url: https://www.bindingdb.org/rwd/bind/chemsearch/marvin/SDFdownload.jsp?all_download=yes
description: "Curated binding affinity datasets for proteinโ€“ligand interaction benchmarking."
tags: [benchmarks-and-datasets]
tasks: []
modalities: []
organism: []
api: false
- id: cancer_therapeutics_response_portal_ctrp
name: "Cancer Therapeutics Response Portal (CTRP)"
type: benchmark
url: https://portals.broadinstitute.org/ctrp/
description: "Drug sensitivity profiles across ~900 cancer cell lines for >400 compounds."
tags: [benchmarks-and-datasets]
tasks: []
modalities: []
organism: []
api: false
- id: clintox
name: "ClinTox"
type: benchmark
url: https://tdcommons.ai/single_pred_tasks/tox/#clintox
description: "Clinical toxicity dataset contrasting FDA-approved drugs with those that failed clinical trials due to toxicity."
tags: [benchmarks-and-datasets]
tasks: []
modalities: []
organism: []
api: false
- id: cptac_clinical_proteomic_tumor_analysis_consortium
name: "CPTAC (Clinical Proteomic Tumor Analysis Consortium)"
type: benchmark
url: https://proteomics.cancer.gov/programs/cptac
description: "Multi-omic proteogenomic datasets for multiple cancer types linking proteomics with genomics."
tags: [benchmarks-and-datasets]
tasks: []
modalities: []
organism: []
api: false
- id: crossdocked2020
name: "CrossDocked2020"
type: benchmark
url: https://arxiv.org/abs/2001.01037
description: "Large-scale dataset for structure-based virtual screening."
tags: [benchmarks-and-datasets]
tasks: []
modalities: []
organism: []
api: false
- id: dud_e_directory_of_useful_decoys_enhanced
name: "DUD-E (Directory of Useful Decoys, Enhanced)"
type: benchmark
url: http://dude.docking.org/
description: "Structure-based virtual screening benchmark with active ligands and challenging decoy sets across diverse protein targets."
tags: [benchmarks-and-datasets]
tasks: []
modalities: []
organism: []
api: false
- id: flip_fitness_landscape_inference_for_proteins
name: "FLIP (Fitness Landscape Inference for Proteins)"
type: benchmark
url: https://github.com/J-SNACKKB/FLIP
description: "Benchmark collection of protein fitness landscape datasets for evaluating protein ML models."
tags: [benchmarks-and-datasets]
tasks: []
modalities: []
organism: []
api: false
- id: guacamol
name: "GuacaMol"
type: benchmark
url: https://github.com/BenevolentAI/guacamol
description: "Benchmark suite for generative molecular design models."
tags: [benchmarks-and-datasets]
tasks: []
modalities: []
organism: []
api: false
- id: hest_xenium_virtual_spatial_transcriptomics
name: "HEST Xenium virtual spatial transcriptomics"
type: benchmark
url: https://huggingface.co/datasets/ratschlab/HEST_Xenium_virtual_spatial_transcriptomics
description: "DeepSpot-M predicted transcriptome-wide ST for 59 HEST-1k 10x Xenium samples (~13.3M cells) (gated). Paper: [DeepSpot-M](https://www.medrxiv.org/content/10.64898/2026.06.19.26356060v1)."
tags: [benchmarks-and-datasets]
tasks: []
modalities: []
organism: []
api: false
- id: jump_cell_painting_datasets
name: "JUMP Cell Painting Datasets"
type: benchmark
url: https://github.com/jump-cellpainting/datasets
description: "Consortium-scale cell imaging perturbation datasets (chemical and genetic) for phenotypic profiling and drug discovery research."
tags: [benchmarks-and-datasets]
tasks: []
modalities: []
organism: []
api: false
- id: lincs_l1000
name: "LINCS L1000"
type: benchmark
url: https://lincsproject.org/LINCS/tools/workflows/find-the-best-place-to-obtain-the-lincs-l1000-data
description: "Gene expression profiles (978 landmark genes) for >20,000 chemical and genetic perturbations across cell lines."
tags: [benchmarks-and-datasets]
tasks: []
modalities: []
organism: []
api: false
- id: moleculenet
name: "MoleculeNet"
type: benchmark
url: http://moleculenet.ai/
description: "Benchmark datasets for molecular machine learning."
tags: [benchmarks-and-datasets]
tasks: []
modalities: []
organism: []
api: false
- id: moses
name: "MOSES"
type: benchmark
url: https://github.com/molecularsets/moses
description: "Benchmarking platform for molecular generation models."
tags: [benchmarks-and-datasets]
tasks: []
modalities: []
organism: []
api: false
- id: ogb_open_graph_benchmark
name: "OGB (Open Graph Benchmark)"
type: benchmark
url: https://ogb.stanford.edu/
description: "Large-scale graph ML benchmark suite including biological datasets such as ogbl-ppa (protein-protein associations) and ogbg-molhiv."
tags: [benchmarks-and-datasets]
tasks: []
modalities: []
organism: []
api: false
- id: openbiolink
name: "OpenBioLink"
type: benchmark
url: https://github.com/OpenBioLink/OpenBioLink
description: "Benchmark datasets for biological knowledge graph completion."
tags: [benchmarks-and-datasets]
tasks: []
modalities: []
organism: []
api: false
- id: pharmgkb
name: "PharmGKB"
type: benchmark
url: https://www.pharmgkb.org/
description: "Curated pharmacogenomics dataset linking genetic variants to drug response phenotypes across thousands of drugs."
tags: [benchmarks-and-datasets]
tasks: []
modalities: []
organism: []
api: false
- id: pk_db
name: "PK-DB"
type: benchmark
url: https://pk-db.com/
description: "Open database of experimental pharmacokinetics (PK) and ADME data from clinical and preclinical studies."
tags: [benchmarks-and-datasets]
tasks: []
modalities: []
organism: []
api: false
- id: prism
name: "PRISM"
type: benchmark
url: https://depmap.org/portal/prism/
description: "Cancer drug sensitivity profiling of >4,500 drugs across >900 cancer cell lines using pooled-cell-line barcoding."
tags: [benchmarks-and-datasets]
tasks: []
modalities: []
organism: []
api: false
- id: proteingym
name: "ProteinGym"
type: benchmark
url: https://github.com/OATML-Markslab/ProteinGym
description: "Large-scale benchmark of deep mutational scanning assays for evaluating protein fitness landscape models."
tags: [benchmarks-and-datasets]
tasks: []
modalities: []
organism: []
api: false
- id: qm9
name: "QM9"
type: benchmark
url: https://figshare.com/collections/Quantum_chemistry_structures_and_properties_of_134_kilo_molecules/978904
description: "Quantum chemistry properties for 134K stable small organic molecules computed at DFT level."
tags: [benchmarks-and-datasets]
tasks: []
modalities: []
organism: []
api: false
- id: scib_single_cell_integration_benchmarks
name: "scIB (Single-cell Integration Benchmarks)"
type: benchmark
url: https://github.com/theislab/scib
description: "Comprehensive benchmarking framework for single-cell data integration methods."
tags: [benchmarks-and-datasets]
tasks: []
modalities: []
organism: []
api: false
- id: scperturb
name: "scPerturb"
type: benchmark
url: https://github.com/sanderlab/scPerturb
description: "Curated and continuously updated single-cell perturbation data resource spanning CRISPR and drug perturbation studies."
tags: [benchmarks-and-datasets]
tasks: []
modalities: []
organism: []
api: false
- id: sider_side_effect_resource
name: "SIDER (Side Effect Resource)"
type: benchmark
url: http://sideeffects.embl.de/
description: "Database of 1,430 approved drugs with their recorded adverse drug reactions across 27 system-organ classes."
tags: [benchmarks-and-datasets]
tasks: []
modalities: []
organism: []
api: false
- id: tabula_muris
name: "Tabula Muris"
type: benchmark
url: https://tabula-muris.ds.czbiohub.org/
description: "Comprehensive single-cell atlas of 20 mouse organs and tissues, enabling cross-tissue and cross-species comparisons."
tags: [benchmarks-and-datasets]
tasks: []
modalities: []
organism: []
api: false
- id: tabula_sapiens
name: "Tabula Sapiens"
type: benchmark
url: https://tabula-sapiens-portal.ds.czbiohub.org/
description: "Comprehensive human single-cell atlas of ~500K cells from 24 organs and tissues across multiple donors."
tags: [benchmarks-and-datasets]
tasks: []
modalities: []
organism: []
api: false
- id: tape_tasks_assessing_protein_embeddings
name: "TAPE (Tasks Assessing Protein Embeddings)"
type: benchmark
url: https://github.com/songlab-cal/tape
description: "Benchmark suite of five biologically meaningful semi-supervised learning tasks for evaluating protein representations."
tags: [benchmarks-and-datasets]
tasks: []
modalities: []
organism: []
api: false
- id: tcga_virtual_spatial_transcriptomics_atlas
name: "TCGA virtual spatial transcriptomics atlas"
type: benchmark
url: https://huggingface.co/datasets/ratschlab/TCGA_virtual_spatial_transcriptomics_atlas
description: "DeepSpot-M predicted transcriptome-wide ST for TCGA H&E (FF + FFPE; 28,664 slides / 32 cancer types; gated). Paper: [DeepSpot-M](https://www.medrxiv.org/content/10.64898/2026.06.19.26356060v1)."
tags: [benchmarks-and-datasets]
tasks: []
modalities: []
organism: []
api: false
- id: the_cancer_genome_atlas_tcga
name: "The Cancer Genome Atlas (TCGA)"
type: benchmark
url: https://www.cancer.gov/about-nci/organization/ccg/research/structural-genomics/tcga
description: "Comprehensive multi-omics (genomics, transcriptomics, proteomics, methylation) dataset for 33 cancer types across ~11,000 patients."
tags: [benchmarks-and-datasets]
tasks: []
modalities: []
organism: []
api: false
- id: therapeutics_data_commons_tdc
name: "Therapeutics Data Commons (TDC)"
type: benchmark
url: https://tdcommons.ai/
description: "Unified benchmark suite covering ADMET, drug-target interaction, drug response, and more."
tags: [benchmarks-and-datasets]
tasks: []
modalities: []
organism: []
api: false
- id: tox21
name: "Tox21"
type: benchmark
url: https://tripod.nih.gov/tox21/challenge/
description: "12,707 compounds tested in 12 nuclear receptor and stress-response pathway biochemical assays for toxicity prediction."
tags: [benchmarks-and-datasets]
tasks: []
modalities: []
organism: []
api: false
- id: uk_biobank
name: "UK Biobank"
type: benchmark
url: https://www.ukbiobank.ac.uk/
description: "Large-scale biomedical database of ~500K participants with genetic, imaging, and health data for population genetics and disease studies."
tags: [benchmarks-and-datasets]
tasks: []
modalities: []
organism: []
api: false
- id: 10x_genomics_dataset
name: "10x Genomics Dataset"
type: database
url: https://www.10xgenomics.com/resources/datasets
description: "Collection of single-cell datasets."
tags: [genome]
tasks: []
modalities: [Genomics]
organism: []
api: false
- id: alphafold_protein_structure_database
name: "AlphaFold Protein Structure Database"
type: database
url: https://alphafold.ebi.ac.uk/api-docs
description: "3D protein structure predictions."
tags: [protein]
tasks: []
modalities: [Protein]
organism: []
api: false
- id: bindingdb
name: "BindingDB"
type: database
url: https://www.bindingdb.org/rwd/bind/index.jsp
description: "Compounds and target database."
tags: [chemical-protein-interaction, interaction]
tasks: []
modalities: [Protein, Small Molecule]
organism: []
api: false
- id: biocyc
name: "BioCyc"
type: database
url: https://biocyc.org/
description: "Collection of pathway/genome databases across thousands of organisms."
tags: [pathway]
tasks: []
modalities: [Pathway]
organism: []
api: false
- id: biogrid
name: "BioGRID"
type: database
url: https://thebiogrid.org/
description: "Protein, genetic, and chemical interactions."
tags: [interaction, protein-protein-interaction]
tasks: []
modalities: [Protein]
organism: []
api: false
- id: cancer_cell_line_encyclopedia
name: "Cancer Cell Line Encyclopedia"
type: database
url: https://sites.broadinstitute.org/ccle/
description: "Database of ~1000 cancer cell lines."
tags: [drug-cell-line-response, interaction]
tasks: []
modalities: [Gene Expression, Small Molecule]
organism: []
api: false
- id: catalogue_of_somatic_mutations_in_cancer_cosmic
name: "Catalogue Of Somatic Mutations In Cancer (COSMIC)"
type: database
url: https://cancer.sanger.ac.uk/cosmic
description: "Resource on somatic mutations in cancers."
tags: [genome]
tasks: []
modalities: [Genomics]
organism: []
api: false
- id: cath_database
name: "CATH database"
type: database
url: https://www.cathdb.info/
description: "Hierarchical classification of protein domain structures."
tags: [protein]
tasks: []
modalities: [Protein]
organism: []
api: false
- id: cbioportal
name: "cBioPortal"
type: database
url: https://www.cbioportal.org/
description: "Cancer genomics database; aggregating many patient datasets."
tags: [genome]
tasks: []
modalities: [Genomics]
organism: []
api: false
- id: cellminer_cross_database_cellminercdb
name: "CellMiner Cross Database (CellMinerCDB)"
type: database
url: https://discover.nci.nih.gov/cellminercdb/
description: "Integrates multiple cancer cell line databases."
tags: [drug-cell-line-response, interaction]
tasks: []
modalities: [Gene Expression, Small Molecule]
organism: []
api: false
- id: chebi
name: "ChEBI"
type: database
url: https://www.ebi.ac.uk/chebi/
description: "Database focused on small chemical compounds."
tags: [compound]
tasks: []
modalities: [Small Molecule]
organism: []
api: false
- id: chembl
name: "ChEMBL"
type: database
url: https://www.ebi.ac.uk/chembl/
description: "Bioactive molecules with drug-like properties."
tags: [compound]
tasks: []
modalities: [Small Molecule]
organism: []
api: false
- id: chemspider
name: "ChemSpider"
type: database
url: http://www.chemspider.com/
description: "Chemical structure database."
tags: [compound]
tasks: []
modalities: [Small Molecule]
organism: []
api: false
- id: clinicaltrials_gov
name: "ClinicalTrials.gov"
type: database
url: https://clinicaltrials.gov/
description: "Privately and publicly funded clinical studies."
tags: [clinical-trial]
tasks: []
modalities: [Clinical]
organism: []
api: false
- id: comparative_toxicogenomics_database
name: "Comparative Toxicogenomics Database"
type: database
url: http://ctdbase.org/
description: "Chemical-gene interactions, chemical-disease and gene-disease associations, chemical-phenotype associations."
tags: [drug-gene-interaction, interaction]
tasks: []
modalities: [Gene, Small Molecule]
organism: []
api: false
- id: critical_assessment_of_structure_prediction_casp
name: "Critical Assessment of Structure Prediction (CASP)"
type: database
url: https://predictioncenter.org/
description: "Assessing methods for protein structure prediction."
tags: [protein]
tasks: []
modalities: [Protein]
organism: []
api: false
- id: cz_cellxgene
name: "CZ CELLxGENE"
type: database
url: https://cellxgene.cziscience.com/
description: "Single-cell dataset repository and interactive explorer from the Chan Zuckerberg Initiative."
tags: [scrna]
tasks: []
modalities: [Single Cell]
organism: []
api: false
- id: davis_kinase_inhibitors_db
name: "Davis kinase inhibitors DB"
type: database
url: http://staff.cs.utu.fi/~aijrinas/dti/
description: "Experimental kinase inhibitor binding affinity dataset for proteinโ€“ligand interaction research."
tags: [chemical-protein-interaction, interaction]
tasks: []
modalities: [Protein, Small Molecule]
organism: []
api: false
- id: dependency_map_depmap
name: "Dependency Map (DepMap)"
type: database
url: https://depmap.org/portal/
description: "CRISPR-Cas9 screens in cancer cell lines."
tags: [genome]
tasks: []
modalities: [Genomics]
organism: []
api: false
- id: dgidb
name: "DGIdb"
type: database
url: https://www.dgidb.org/
description: "Drug-gene interactions and the druggable genome."
tags: [drug-gene-interaction, interaction]
tasks: []
modalities: [Gene, Small Molecule]
organism: []
api: false
- id: diseases
name: "DISEASES"
type: database
url: https://diseases.jensenlab.org/
description: "Geneโ€“disease association database integrating evidence from text mining, curated databases, and experimental data."
tags: [disease]
tasks: []
modalities: [Disease]
organism: []
api: false
- id: disgenet
name: "DisGeNET"
type: database
url: https://www.disgenet.org/
description: "Database of gene-disease associations integrating expert-curated and GWAS data."
tags: [disease]
tasks: []
modalities: [Disease]
organism: []
api: false
- id: drkg
name: "DRKG"
type: database
url: https://github.com/gnn4dr/DRKG
description: "Large-scale biological knowledge graph for drug discovery."
tags: [interaction, knowledge-graph]
tasks: []
modalities: [Knowledge Graph]
organism: []
api: false
- id: drug_mechanism_database_drugmechdb
name: "Drug Mechanism Database (DrugMechDB)"
type: database
url: https://github.com/SuLab/DrugMechDB/tree/2.0.1
description: "Mechanisms of action from drug to disease."
tags: [interaction, knowledge-graph]
tasks: []
modalities: [Knowledge Graph]
organism: []
api: false
- id: drug_repurposing_hub
name: "Drug Repurposing Hub"
type: database
url: https://repo-hub.broadinstitute.org/repurposing#download-data
description: "Collections of drug repurposing data (drug, MoA, target, etc)."
tags: [compound]
tasks: []
modalities: [Small Molecule]
organism: []
api: false
- id: drugbank
name: "DrugBank"
type: database
url: https://go.drugbank.com/
description: "Database of drugs and targets (University of Alberta)."
tags: [disease]
tasks: []
modalities: [Disease]
organism: []
api: false
- id: drugcentral
name: "DrugCentral"
type: database
url: http://drugcentral.org/
description: "Online drug compendium with drug mode of action and indication information."
tags: [compound]
tasks: []
modalities: [Small Molecule]
organism: []
api: false
- id: drugtargetcommons
name: "DrugTargetCommons"
type: database
url: https://drugtargetcommons.fimm.fi/
description: "Community platform for curating and integrating experimental bioactivity data across drugs and targets."
tags: [compound]
tasks: []
modalities: [Small Molecule]
organism: []
api: false
- id: encode
name: "ENCODE"
type: database
url: https://www.encodeproject.org/
description: "Encyclopedia of DNA Elements; regulatory and functional genomic elements across the genome."
tags: [genome]
tasks: []
modalities: [Genomics]
organism: []
api: false
- id: ensembl
name: "Ensembl"
type: database
url: https://www.ensembl.org/
description: "Genome browser and annotation database for vertebrate and other eukaryotic genomes."
tags: [genome]
tasks: []
modalities: [Genomics]
organism: []
api: false
- id: eu_drug_regulating_authorities_clinical_trials_db_eudract
name: "EU Drug Regulating Authorities Clinical Trials DB (EudraCT)"
type: database
url: https://eudract.ema.europa.eu/
description: "European clinical trial database."
tags: [clinical-trial]
tasks: []
modalities: [Clinical]
organism: []
api: false
- id: fantom5
name: "FANTOM5"
type: database
url: https://fantom.gsc.riken.jp/5/
description: "Functional annotation of mammalian genome; comprehensive atlas of active enhancers, promoters, and transcription start sites across human and mouse cell types."
tags: [genome]
tasks: []
modalities: [Genomics]
organism: []
api: false
- id: genbank
name: "GenBank"
type: database
url: https://www.ncbi.nlm.nih.gov/genbank/
description: "NCBI's database of genetic sequences."
tags: [genome]
tasks: []
modalities: [Genomics]
organism: []
api: false
- id: gene_expression_omnibus
name: "Gene Expression Omnibus"
type: database
url: https://www.ncbi.nlm.nih.gov/geo/
description: "Public functional genomics database."
tags: [scrna]
tasks: []
modalities: [Single Cell]
organism: []
api: false
- id: genomics_of_drug_sensitivity_in_cancer_gdsc
name: "Genomics of Drug Sensitivity in Cancer (GDSC)"
type: database
url: https://www.cancerrxgene.org/
description: "Drug sensitivity for ~1000 human cancer cell lines and hundreds of compounds."
tags: [benchmarks-and-datasets, drug-cell-line-response, interaction]
tasks: []
modalities: [Gene Expression, Small Molecule]
organism: []
api: false
- id: gnomad
name: "gnomAD"
type: database
url: https://gnomad.broadinstitute.org/
description: "Genome Aggregation Database; genetic variation from large-scale sequencing projects."
tags: [genome]
tasks: []
modalities: [Genomics]
organism: []
api: false
- id: hetionet
name: "Hetionet"
type: database
url: https://github.com/hetio/hetionet
description: "Heterogeneous network integrating genes, diseases, drugs, pathways, and more."
tags: [interaction, knowledge-graph]
tasks: []
modalities: [Knowledge Graph]
organism: []
api: false
- id: hippie
name: "HIPPIE"
type: database
url: http://cbdm-01.zdv.uni-mainz.de/~mschaefer/hippie/
description: "Human protein-protein interaction database."
tags: [interaction, protein-protein-interaction]
tasks: []
modalities: [Protein]
organism: []
api: false
- id: hmdb_human_metabolome_database
name: "HMDB (Human Metabolome Database)"
type: database
url: https://hmdb.ca/
description: "Comprehensive database of small molecule metabolites found in the human body."
tags: [compound]
tasks: []
modalities: [Small Molecule]
organism: []
api: false
- id: human_cell_atlas
name: "Human Cell Atlas"
type: database
url: https://www.humancellatlas.org/
description: "Open global atlas of all cells in the human body."
tags: [scrna]
tasks: []
modalities: [Single Cell]
organism: []
api: false
- id: human_genome_resources_at_ncbi
name: "Human Genome Resources at NCBI"
type: database
url: https://www.ncbi.nlm.nih.gov/projects/genome/guide/human/index.shtml
description: "Database for genomics, proteomics, transcriptomics, and systems biology."
tags: [genome]
tasks: []
modalities: [Genomics]
organism: []
api: false
- id: human_phenotype_ontology_hpo
name: "Human Phenotype Ontology (HPO)"
type: database
url: https://hpo.jax.org/
description: "Standardized vocabulary of phenotypic abnormalities in human disease, linking genes, variants, and clinical features."
tags: [disease]
tasks: []
modalities: [Disease]
organism: []
api: false
- id: icd10
name: "ICD10"
type: database
url: https://icd.who.int/browse10/2019/en
description: "International Classification of Diseases, 10th revision."
tags: [clinical-trial]
tasks: []
modalities: [Clinical]
organism: []
api: false
- id: intact
name: "IntAct"
type: database
url: https://www.ebi.ac.uk/intact/home
description: "Open-source molecular interaction database and analysis system from EMBL-EBI."
tags: [interaction, protein-protein-interaction]
tasks: []
modalities: [Protein]
organism: []
api: false
- id: interpro
name: "InterPro"
type: database
url: https://www.ebi.ac.uk/interpro/
description: "Protein families, domains, and functional sites database integrating 14 member databases including Pfam and PROSITE."
tags: [protein]
tasks: []
modalities: [Protein]
organism: []
api: false
- id: jaspar
name: "JASPAR"
type: database
url: http://jaspar.genereg.net/
description: "Database of transcription factor binding profiles."
tags: [genome]
tasks: []
modalities: [Genomics]
organism: []
api: false
- id: kegg_compound
name: "KEGG COMPOUND"
type: database
url: https://www.genome.jp/kegg/compound/
description: "Collection of small molecules and biopolymers."
tags: [compound]
tasks: []
modalities: [Small Molecule]
organism: []
api: false
- id: kegg_drug
name: "KEGG DRUG"
type: database
url: https://www.genome.jp/kegg/drug/
description: "Comprehensive, approved drug information."
tags: [disease]
tasks: []
modalities: [Disease]
organism: []
api: false
- id: kegg_pathway
name: "KEGG PATHWAY"
type: database
url: https://www.genome.jp/kegg/pathway.html
description: "Collection of pathway maps."
tags: [pathway]
tasks: []
modalities: [Pathway]
organism: []
api: false
- id: kinase_inhibitor_bioactivity_data_kiba
name: "Kinase Inhibitor Bioactivity Data (KIBA)"
type: database
url: https://janeliascicomp.github.io/KIBA/
description: "Integrated bioactivity scores for kinase inhibitors combining Ki, Kd, and IC50 measurements."
tags: [chemical-protein-interaction, interaction]
tasks: []
modalities: [Protein, Small Molecule]
organism: []
api: false
- id: lipid_maps
name: "LIPID MAPS"
type: database
url: https://www.lipidmaps.org/databases/lmsd/overview
description: "Database of lipids."
tags: [compound]
tasks: []
modalities: [Small Molecule]
organism: []
api: false
- id: massbank
name: "MassBank"
type: database
url: http://www.massbank.jp/
description: "Open source databases and tools for mass spectrometry reference spectra."
tags: [mass-spectra]
tasks: []
modalities: [Mass Spectra]
organism: []
api: false
- id: mgnify
name: "MGnify"
type: database
url: https://www.ebi.ac.uk/metagenomics/
description: "Resource for metagenomic and metatranscriptomic data."
tags: [genome]
tasks: []
modalities: [Genomics]
organism: []
api: false
- id: mimic_iv
name: "MIMIC-IV"
type: database
url: https://mimic.mit.edu/
description: "Freely accessible critical care database."
tags: [clinical-trial]
tasks: []
modalities: [Clinical]
organism: []
api: false
- id: mirbase
name: "miRBase"
type: database
url: https://www.mirbase.org/
description: "Reference repository for microRNA gene annotations, sequences, and experimentally validated targets."
tags: [gene-regulatory-network, interaction]
tasks: []
modalities: [Gene Expression]
organism: []
api: false
- id: mona_massbank_of_north_america
name: "MoNA MassBank of North America"
type: database
url: https://mona.fiehnlab.ucdavis.edu/
description: "Meta-database of metabolite mass spectra, metadata, and associated compounds."
tags: [mass-spectra]
tasks: []
modalities: [Mass Spectra]
organism: []
api: false
- id: msigdb_molecular_signatures_database
name: "MSigDB (Molecular Signatures Database)"
type: database
url: https://www.gsea-msigdb.org/gsea/msigdb
description: "Curated gene sets derived from pathways and biological processes."
tags: [pathway]
tasks: []
modalities: [Pathway]
organism: []
api: false
- id: nci60
name: "NCI60"
type: database
url: https://dtp.cancer.gov/discovery_development/nci-60/
description: "Focuses on 60 cancer cell lines and many drugs."
tags: [benchmarks-and-datasets, drug-cell-line-response, interaction]
tasks: []
modalities: [Gene Expression, Small Molecule]
organism: []
api: false
- id: nextprot
name: "NeXtProt"
type: database
url: https://www.nextprot.org/
description: "Expert knowledge base on human proteins with deep functional annotation, complementary to UniProt."
tags: [protein]
tasks: []
modalities: [Protein]
organism: []
api: false
- id: oadb_observed_antibody_space_database
name: "OADB (Observed Antibody Space Database)"
type: database
url: http://opig.stats.ox.ac.uk/webapps/oas/
description: "Database of antibody sequences from immune repertoire sequencing."
tags: [protein]
tasks: []
modalities: [Protein]
organism: []
api: false
- id: omim_online_mendelian_inheritance_in_man
name: "OMIM (Online Mendelian Inheritance in Man)"
type: database
url: https://www.omim.org/
description: "Comprehensive database of human genes and genetic disorders."
tags: [disease]
tasks: []
modalities: [Disease]
organism: []
api: false
- id: omnipath
name: "OmniPath"
type: database
url: https://omnipathdb.org/
description: "Comprehensive resource integrating protein interactions, signaling pathways, gene regulatory networks, and miRNA targets from over 100 databases."
tags: [pathway]
tasks: []
modalities: [Pathway]
organism: []
api: false
- id: oncokb
name: "OncoKB"
type: database
url: https://www.oncokb.org/
description: "Precision oncology knowledge base of cancer genes, variants, and therapeutic implications."
tags: [genome]
tasks: []
modalities: [Genomics]
organism: []
api: false
- id: open_targets_platform
name: "Open Targets Platform"
type: database
url: https://platform.opentargets.org/
description: "Systematic target identification and prioritization platform integrating genetics, genomics, and drug data for drug discovery."
tags: [disease]
tasks: []
modalities: [Disease]
organism: []
api: false
- id: pathwaycommons
name: "PathwayCommons"
type: database
url: https://www.pathwaycommons.org/
description: "Database of pathways and interactions."
tags: [pathway]
tasks: []
modalities: [Pathway]
organism: []
api: false
- id: pdbbind
name: "PDBBind"
type: database
url: https://www.pdbbind-plus.org.cn/
description: "Binding affinity data for biomolecular complexes."
tags: [chemical-protein-interaction, interaction]
tasks: []
modalities: [Protein, Small Molecule]
organism: []
api: false
- id: pfam
name: "Pfam"
type: database
url: https://www.ebi.ac.uk/interpro/entry/pfam/
description: "Database of protein families described by multiple sequence alignments and hidden Markov models."
tags: [protein]
tasks: []
modalities: [Protein]
organism: []
api: false
- id: primekg
name: "PrimeKG"
type: database
url: https://github.com/mims-harvard/PrimeKG
description: "Multi-modal precision medicine knowledge graph integrating clinical, genetic, and drug data."
tags: [interaction, knowledge-graph]
tasks: []
modalities: [Knowledge Graph]
organism: []
api: false
- id: protein_data_bank_pdb
name: "PROTEIN DATA BANK (PDB)"
type: database
url: https://www.rcsb.org/
description: "3D structures of proteins, nucleic acids, complexes."
tags: [protein]
tasks: []
modalities: [Protein]
organism: []
api: false
- id: pubchem
name: "PubChem"
type: database
url: https://pubchem.ncbi.nlm.nih.gov/
description: "One of the largest chemical databases (compounds, genes, and proteins)."
tags: [compound]
tasks: []
modalities: [Small Molecule]
organism: []
api: false
- id: rcsb_protein_data_bank
name: "RCSB Protein Data Bank"
type: database
url: https://www.rcsb.org/
description: "Repository for structural data of biological molecules."
tags: [protein]
tasks: []
modalities: [Protein]
organism: []
api: false
- id: reactome
name: "Reactome"
type: database
url: https://reactome.org/
description: "Expert-curated, peer-reviewed pathway database with detailed reaction mechanisms."
tags: [pathway]
tasks: []
modalities: [Pathway]
organism: []
api: false
- id: regnetwork
name: "RegNetwork"
type: database
url: http://www.regnetworkweb.org/
description: "Database of gene regulatory networks covering transcription factorโ€“target gene and miRNAโ€“gene interaction data across multiple species."
tags: [gene-regulatory-network, interaction]
tasks: []
modalities: [Gene Expression]
organism: []
api: false
- id: rfam
name: "Rfam"
type: database
url: https://rfam.org/
description: "Database of RNA families with sequence alignments and consensus structures."
tags: [genome]
tasks: []
modalities: [Genomics]
organism: []
api: false
- id: rhea
name: "Rhea"
type: database
url: https://www.rhea-db.org/
description: "Database of chemical reactions."
tags: [compound]
tasks: []
modalities: [Small Molecule]
organism: []
api: false
- id: roadmap_epigenomics
name: "ROADMAP Epigenomics"
type: database
url: http://www.roadmapepigenomics.org/
description: "Reference epigenome maps for 111 primary human cell types and tissues, including histone modifications, chromatin accessibility, and DNA methylation."
tags: [genome]
tasks: []
modalities: [Genomics]
organism: []
api: false
- id: sabdab
name: "SAbDab"
type: database
url: https://opig.stats.ox.ac.uk/webapps/sabdab-sabpred/sabdab
description: "Structural Antibody Database containing all antibody structures in the PDB."
tags: [protein]
tasks: []
modalities: [Protein]
organism: []
api: false
- id: signor_2_0
name: "SIGNOR 2.0"
type: database
url: https://signor.uniroma2.it/
description: "Database of causal signaling interactions and pathways, with signed and directed relationships between proteins."
tags: [pathway]
tasks: []
modalities: [Pathway]
organism: []
api: false
- id: single_cell_expression_atlas
name: "Single Cell Expression Atlas"
type: database
url: https://www.ebi.ac.uk/gxa/sc/home
description: "Public database for single-cell RNA."
tags: [scrna]
tasks: []
modalities: [Single Cell]
organism: []
api: false
- id: single_cell_portal
name: "Single Cell PORTAL"
type: database
url: https://singlecell.broadinstitute.org/single_cell
description: "Public database for single-cell RNA."
tags: [scrna]
tasks: []
modalities: [Single Cell]
organism: []
api: false
- id: snap
name: "SNAP"
type: database
url: https://snap.stanford.edu/biodata/datasets/10002/10002-ChG-Miner.html
description: "Dataset of drug-gene interactions."
tags: [drug-gene-interaction, interaction]
tasks: []
modalities: [Gene, Small Molecule]
organism: []
api: false
- id: stitch
name: "STITCH"
type: database
url: http://stitch.embl.de/
description: "Chemical-protein interactions."
tags: [chemical-protein-interaction, interaction]
tasks: []
modalities: [Protein, Small Molecule]
organism: []
api: false
- id: string
name: "STRING"
type: database
url: https://string-db.org/
description: "PPI networks for multiple organisms."
tags: [interaction, protein-protein-interaction]
tasks: []
modalities: [Protein]
organism: []
api: false
- id: the_genotype_tissue_expression_gtex
name: "The Genotype-Tissue Expression (GTEx)"
type: database
url: https://gtexportal.org/home/
description: "Human gene expression and regulation resource."
tags: [genome]
tasks: []
modalities: [Genomics]
organism: []
api: false
- id: the_human_protein_atlas
name: "THE HUMAN PROTEIN ATLAS"
type: database
url: https://www.proteinatlas.org/
description: "Comprehensive human protein database (cells, tissues, organs)."
tags: [protein]
tasks: []
modalities: [Protein]
organism: []
api: false
- id: therapeutic_target_database
name: "Therapeutic Target Database"
type: database
url: https://idrblab.net/ttd/full-data-download
description: "Drug-target, target-disease, and drug-disease datasets."
tags: [compound]
tasks: []
modalities: [Small Molecule]
organism: []
api: false
- id: trrust_v2
name: "TRRUST v2"
type: database
url: https://www.grnpedia.org/trrust/
description: "Manually curated database of human and mouse transcriptional regulatory interactions between transcription factors and their target genes, expanded with literature-derived evidence."
tags: [gene-regulatory-network, interaction]
tasks: []
modalities: [Gene Expression]
organism: []
api: false
- id: ucsc_genome_browser
name: "UCSC Genome Browser"
type: database
url: https://genome.ucsc.edu/
description: "UCSC's genome browser."
tags: [genome]
tasks: []
modalities: [Genomics]
organism: []
api: false
- id: uniclust
name: "Uniclust"
type: database
url: https://uniclust.mmseqs.com/
description: "Clustered protein sequence databases."
tags: [protein]
tasks: []
modalities: [Protein]
organism: []
api: false
- id: uniprot
name: "UniProt"
type: database
url: https://www.uniprot.org/
description: "Functional information on proteins."
tags: [protein]
tasks: []
modalities: [Protein]
organism: []
api: false
- id: uniref
name: "UniRef"
type: database
url: https://www.uniprot.org/uniref/
description: "Non-redundant sequence database clustering UniProtKB entries at multiple sequence identity thresholds."
tags: [protein]
tasks: []
modalities: [Protein]
organism: []
api: false
- id: wikipathways
name: "WikiPathways"
type: database
url: https://wikipathways.org/
description: "Database of biological pathways."
tags: [pathway]
tasks: []
modalities: [Pathway]
organism: []
api: false
- id: zinc_ligand_discovery_database
name: "ZINC ligand discovery database"
type: database
url: https://zinc.docking.org/
description: "Free database of commercially-available compounds for virtual screening."
tags: [compound]
tasks: []
modalities: [Small Molecule]
organism: []
api: false
- id: aestetik
name: "AESTETIK"
type: model
url: https://github.com/ratschlab/aestetik
description: "Autoencoder for spatial transcriptomics representation learning using topology and histology image knowledge."
tags: [foundation-models, single-cell-foundation-models, spatial-foundation-models]
tasks: [Foundation Model]
modalities: [Single Cell, Spatial Transcriptomics]
organism: []
api: false
- id: ai4chem_chemllm_7b_chat
name: "AI4Chem/ChemLLM-7B-Chat"
type: model
url: https://huggingface.co/AI4Chem/ChemLLM-7B-Chat
description: "LLM for chemical & molecular science."
tags: [llm-for-biology]
tasks: [Language Modeling]
modalities: [Text]
organism: []
api: false
- id: alphafold3
name: "AlphaFold3"
type: model
url: https://github.com/google-deepmind/alphafold3
description: "Predicts structures of proteins, nucleic acids, small molecules, and their complexes."
tags: [foundation-models, protein-foundation-models, protein-structure-prediction-and-design]
tasks: [Foundation Model, Protein Structure Prediction]
modalities: [Protein]
organism: []
api: false
- id: ankh
name: "Ankh"
type: model
url: https://github.com/agemagician/Ankh
description: "Efficient protein language model optimized for downstream prediction tasks including secondary structure, localization, and function annotation."
tags: [foundation-models, pre-trained-embedding, protein-foundation-models]
tasks: [Foundation Model]
modalities: [Protein]
organism: []
api: false
- id: babel
name: "BABEL"
type: model
url: https://github.com/wukevin/babel
description: "Cross-modality translation model enabling prediction between scRNA-seq and scATAC-seq profiles without requiring paired single-cell measurements."
tags: [foundation-models, multi-omics-foundation-models, single-cell-foundation-models]
tasks: [Foundation Model]
modalities: [Multi-Omics, Single Cell]
organism: []
api: false
- id: basenji
name: "Basenji"
type: model
url: https://github.com/calico/basenji
description: "Sequential regulatory activity prediction from DNA sequences."
tags: [foundation-models, genomics-foundation-models]
tasks: [Foundation Model]
modalities: [Genomics]
organism: []
api: false
- id: biogpt
name: "BioGPT"
type: model
url: https://github.com/microsoft/BioGPT
description: "LLM for biomedical text generation."
tags: [llm-for-biology]
tasks: [Language Modeling]
modalities: [Text]
organism: []
api: false
- id: biomedclip
name: "BiomedCLIP"
type: model
url: https://huggingface.co/microsoft/BiomedCLIP-PubMedBERT_256-vit_g_14
description: "CLIP-based vision-language foundation model for biomedical images and text trained on PubMed figureโ€“caption pairs."
tags: [foundation-models, multi-modal-foundation-models]
tasks: [Foundation Model]
modalities: [Multi-Modal]
organism: []
api: false
- id: biomedlm
name: "BioMedLM"
type: model
url: https://huggingface.co/stanford-crfm/BioMedLM
description: "2.7B parameter GPT-2-style language model trained exclusively on biomedical literature from PubMed for biomedical question answering and text generation."
tags: [llm-for-biology]
tasks: [Language Modeling]
modalities: [Text]
organism: []
api: false
- id: boltz_1
name: "Boltz-1"
type: model
url: https://github.com/jwohlwend/boltz
description: "Open-source all-atom biomolecular structure prediction model for proteins, nucleic acids, small molecules, and their complexes achieving AlphaFold3-level accuracy."
tags: [foundation-models, protein-foundation-models, protein-structure-prediction-and-design]
tasks: [Foundation Model, Protein Structure Prediction]
modalities: [Protein]
organism: []
api: false
- id: borzoi
name: "Borzoi"
type: model
url: https://github.com/calico/borzoi
description: "Extended successor to Enformer for predicting RNA-seq coverage from long genomic sequence windows (524 kb) with improved resolution."
tags: [foundation-models, genomics-foundation-models]
tasks: [Foundation Model]
modalities: [Genomics]
organism: []
api: false
- id: bulkformer
name: "BulkFormer"
type: model
url: https://github.com/KangBoming/BulkFormer
description: "Foundation model for bulk RNA-seq data; learns general transcriptomic representations."
tags: [foundation-models, single-cell-foundation-models, transcriptomics-foundation-models]
tasks: [Foundation Model]
modalities: [Single Cell, Transcriptomics]
organism: []
api: false
- id: caduceus
name: "Caduceus"
type: model
url: https://github.com/kuleshov-group/caduceus
description: "Bidirectional equivariant long-range DNA sequence model based on Mamba."
tags: [foundation-models, genomics-foundation-models]
tasks: [Foundation Model]
modalities: [Genomics]
organism: []
api: false
- id: cancerfoundation
name: "CancerFoundation"
type: model
url: https://github.com/BoevaLab/CancerFoundation
description: "Single-cell RNA-seq foundation model trained exclusively on a curated dataset of malignant cells to learn cancer-specific embeddings."
tags: [foundation-models, single-cell-foundation-models, transcriptomics-foundation-models]
tasks: [Foundation Model]
modalities: [Single Cell, Transcriptomics]
organism: []
api: false
- id: cassia
name: "CASSIA"
type: model
url: https://github.com/ElliotXie/CASSIA
description: "Multi-agent LLM for reference-free, interpretable cell-type annotation of single-cell RNA-seq data, with dedicated annotation, validation, scoring, and reporting agents."
tags: [llm-for-biology]
tasks: [Language Modeling]
modalities: [Text]
organism: []
api: false
- id: cellot
name: "CellOT"
type: model
url: https://github.com/bunnech/cellot
description: "Neural optimal transport framework for predicting single-cell responses to drug and genetic perturbations."
tags: [drug-discovery, drug-perturbation]
tasks: [Drug Discovery, Drug Perturbation]
modalities: [Small Molecule]
organism: []
api: false
- id: cellplm
name: "CellPLM"
type: model
url: https://github.com/OmicsML/CellPLM
description: "Cell pre-trained language model with inter-cell transformer architecture for diverse single-cell analysis tasks."
tags: [foundation-models, single-cell-foundation-models, transcriptomics-foundation-models]
tasks: [Foundation Model]
modalities: [Single Cell, Transcriptomics]
organism: []
api: false
- id: chai_1
name: "Chai-1"
type: model
url: https://github.com/chaidiscovery/chai-lab
description: "Unified molecular structure prediction model covering proteins, nucleic acids, small molecules, and complexes."
tags: [foundation-models, protein-foundation-models, protein-structure-prediction-and-design]
tasks: [Foundation Model, Protein Structure Prediction]
modalities: [Protein]
organism: []
api: false
- id: chatdrug
name: "ChatDrug"
type: model
url: https://github.com/chao1224/ChatDrug
description: "LLM-based conversational pipeline for drug discovery, using natural language prompts for iterative drug editing and optimization."
tags: [llm-for-biology]
tasks: [Language Modeling]
modalities: [Text]
organism: []
api: false
- id: chemberta_2
name: "ChemBERTa-2"
type: model
url: https://github.com/seyonechithrananda/bert-loves-chemistry
description: "RoBERTa-based molecular language model pretrained on SMILES for small-molecule representation learning."
tags: [compound-embedding, compound-foundation-models, foundation-models]
tasks: [Foundation Model]
modalities: [Small Molecule]
organism: []
api: false
- id: chemcpa
name: "chemCPA"
type: model
url: https://github.com/theislab/chemCPA
description: "Compositional perturbation autoencoder for predicting single-cell transcriptional responses to unseen drug perturbations and dose combinations."
tags: [drug-discovery, drug-perturbation]
tasks: [Drug Discovery, Drug Perturbation]
modalities: [Small Molecule]
organism: []
api: false
- id: chief
name: "CHIEF"
type: model
url: https://github.com/hms-dbmi/CHIEF
description: "Clinical Histopathology Imaging Evaluation Foundation model integrating histology images and clinical context for pan-cancer analysis."
tags: [foundation-models, multi-modal-foundation-models]
tasks: [Foundation Model]
modalities: [Multi-Modal]
organism: []
api: false
- id: clawbio
name: "ClawBio"
type: model
url: https://github.com/ClawBio/ClawBio
description: "Bioinformatics-native AI agent skill library with local-first pharmacogenomics, ancestry PCA, semantic similarity, nutrigenomics, and metagenomics skills."
tags: [llm-for-biology]
tasks: [Language Modeling]
modalities: [Text]
organism: []
api: false
- id: cmonge
name: "CMonge"
type: model
url: https://github.com/AI4SCR/conditional-monge-gap
description: "Conditional optimal transport model for generalizable single-cell perturbation response prediction across drugs and doses."
tags: [drug-discovery, drug-perturbation]
tasks: [Drug Discovery, Drug Perturbation]
modalities: [Small Molecule]
organism: []
api: false
- id: concerto
name: "Concerto"
type: model
url: https://github.com/melobio/Concerto-reproducibility
description: "Contrastive self-supervised learning framework for single-cell multimodal data integration, batch correction, and reference-query mapping."
tags: [foundation-models, multi-omics-foundation-models, single-cell-foundation-models]
tasks: [Foundation Model]
modalities: [Multi-Omics, Single Cell]
organism: []
api: false
- id: conch
name: "CONCH"
type: model
url: https://github.com/mahmoodlab/CONCH
description: "Vision-language foundation model for computational pathology trained with contrastive captioning on pathology imageโ€“text pairs."
tags: [foundation-models, single-cell-foundation-models, spatial-foundation-models]
tasks: [Foundation Model]
modalities: [Single Cell, Spatial Transcriptomics]
organism: []
api: false
- id: cyclecdr
name: "cycleCDR"
type: model
url: https://github.com/hliulab/cycleCDR
description: "Interpretable cycle-consistency framework for modeling cellular responses to drug perturbations."
tags: [drug-discovery, drug-perturbation]
tasks: [Drug Discovery, Drug Perturbation]
modalities: [Small Molecule]
organism: []
api: false
- id: deepaeg
name: "DeepAEG"
type: model
url: https://github.com/zhejiangzhuque/DeepAEG
description: "GNN embedding + attention mechanism."
tags: [drug-discovery, drug-response-prediction]
tasks: [Drug Discovery, Drug Response Prediction]
modalities: [Small Molecule]
organism: []
api: false
- id: deepdsc
name: "DeepDSC"
type: model
url: https://ieeexplore-ieee-org.ezp2.lib.umn.edu/stamp/stamp.jsp?tp=&arnumber=8723620&tag=1
description: "Autoencoder + fully connected NN."
tags: [drug-discovery, drug-response-prediction]
tasks: [Drug Discovery, Drug Response Prediction]
modalities: [Small Molecule]
organism: []
api: false
- id: deepdta
name: "DeepDTA"
type: model
url: https://github.com/hkmztrk/DeepDTA
description: "Deep learning model using CNNs on protein sequences and drug SMILES."
tags: [drug-discovery, drug-target-interaction]
tasks: [Drug Discovery, Drug Target Interaction]
modalities: [Protein, Small Molecule]
organism: []
api: false
- id: deeppurpose
name: "DeepPurpose"
type: model
url: https://github.com/kexinhuang12345/DeepPurpose
description: "Deep learning library for drug repurposing."
tags: [drug-discovery, drug-repurposing]
tasks: [Drug Discovery, Drug Repurposing]
modalities: [Small Molecule]
organism: []
api: false
- id: deepsea
name: "DeepSEA"
type: model
url: http://deepsea.princeton.edu/
description: "Deep learning framework for predicting chromatin effects of sequence alterations with single-nucleotide sensitivity across thousands of chromatin features."
tags: [foundation-models, genomics-foundation-models]
tasks: [Foundation Model]
modalities: [Genomics]
organism: []
api: false
- id: deepspot
name: "DeepSpot"
type: model
url: https://github.com/ratschlab/DeepSpot
description: "Deep learning model predicting spatial transcriptomics from H&E images at spot and single-cell resolution."
tags: [foundation-models, single-cell-foundation-models, spatial-foundation-models]
tasks: [Foundation Model]
modalities: [Single Cell, Spatial Transcriptomics]
organism: []
api: false
- id: deepspot_m
name: "DeepSpot-M"
type: model
url: https://github.com/ratschlab/DeepSpotM
description: "Multimodal foundation model for transcriptome-wide virtual spatial transcriptomics from histology."
tags: [foundation-models, single-cell-foundation-models, spatial-foundation-models]
tasks: [Foundation Model]
modalities: [Single Cell, Spatial Transcriptomics]
organism: []
api: false
- id: deepspot2cell
name: "DeepSpot2Cell"
type: model
url: https://github.com/ratschlab/DeepSpot2Cell
description: "Predicts virtual single-cell spatial transcriptomics from H&E using spot-level supervision (NeurIPS 2025 Imageomics)."
tags: [foundation-models, single-cell-foundation-models, spatial-foundation-models]
tasks: [Foundation Model]
modalities: [Single Cell, Spatial Transcriptomics]
organism: []
api: false
- id: dgdrp
name: "DGDRP"
type: model
url: https://github.com/minwoopak/heteronet
description: "Multi-view embedding neural network."
tags: [drug-discovery, drug-response-prediction]
tasks: [Drug Discovery, Drug Response Prediction]
modalities: [Small Molecule]
organism: []
api: false
- id: diffdock
name: "DiffDock"
type: model
url: https://github.com/gcorso/DiffDock
description: "Diffusion generative model for molecular docking, predicting the binding pose of small molecules to protein targets."
tags: [drug-discovery, molecular-generation]
tasks: [Drug Discovery, Molecular Generation]
modalities: [Small Molecule]
organism: []
api: false
- id: diffsbdd
name: "DiffSBDD"
type: model
url: https://github.com/arneschneuing/DiffSBDD
description: "Equivariant diffusion model for structure-based drug design that generates molecules and binding conformations for protein targets."
tags: [drug-discovery, molecular-generation]
tasks: [Drug Discovery, Molecular Generation]
modalities: [Small Molecule]
organism: []
api: false
- id: dnabert
name: "DNABERT"
type: model
url: https://github.com/jerryji1993/DNABERT
description: "Pre-trained bidirectional encoder for DNA sequence analysis."
tags: [foundation-models, genomics-foundation-models]
tasks: [Foundation Model]
modalities: [Genomics]
organism: []
api: false
- id: dnabert_2
name: "DNABERT-2"
type: model
url: https://github.com/Zhihan1996/DNABERT_2
description: "Improved genome foundation model with efficient tokenization."
tags: [foundation-models, genomics-foundation-models]
tasks: [Foundation Model]
modalities: [Genomics]
organism: []
api: false
- id: drgat
name: "drGAT"
type: model
url: https://github.com/inoue0426/drGAT
description: "Attention-based model for drug response prediction with gene explainability."
tags: [drug-discovery, drug-response-prediction]
tasks: [Drug Discovery, Drug Response Prediction]
modalities: [Small Molecule]
organism: []
api: false
- id: drugban
name: "DrugBAN"
type: model
url: https://github.com/peizhenbai/DrugBAN
description: "Bilinear attention network for interpretable DTI prediction."
tags: [drug-discovery, drug-target-interaction]
tasks: [Drug Discovery, Drug Target Interaction]
modalities: [Protein, Small Molecule]
organism: []
api: false
- id: druml
name: "DRUML"
type: model
url: https://github.com/CutillasLab/DRUMLR
description: "Ensemble machine learning framework combining standard ML with deep learning to systematically rank anti-cancer drugs from proteomics and RNA-seq data."
tags: [drug-discovery, drug-response-prediction]
tasks: [Drug Discovery, Drug Response Prediction]
modalities: [Small Molecule]
organism: []
api: false
- id: dtinet
name: "DTINet"
type: model
url: https://github.com/luoyunan/DTINet
description: "Network-based framework integrating heterogeneous biological data for DTI prediction."
tags: [drug-discovery, drug-target-interaction]
tasks: [Drug Discovery, Drug Target Interaction]
modalities: [Protein, Small Molecule]
organism: []
api: false
- id: enformer
name: "Enformer"
type: model
url: https://github.com/deepmind/deepmind-research/tree/master/enformer
description: "Transformer model predicting gene expression from DNA sequence."
tags: [foundation-models, genomics-foundation-models]
tasks: [Foundation Model]
modalities: [Genomics]
organism: []
api: false
- id: esm3
name: "ESM3"
type: model
url: https://github.com/evolutionaryscale/esm
description: "Multimodal protein language model that jointly reasons over sequence, structure, and function for generative protein design and engineering."
tags: [foundation-models, protein-foundation-models, protein-structure-prediction-and-design]
tasks: [Foundation Model, Protein Structure Prediction]
modalities: [Protein]
organism: []
api: false
- id: esmfold
name: "ESMFold"
type: model
url: https://github.com/facebookresearch/esm
description: "Fast protein structure prediction using language model embeddings."
tags: [foundation-models, protein-foundation-models, protein-structure-prediction-and-design]
tasks: [Foundation Model, Protein Structure Prediction]
modalities: [Protein]
organism: []
api: false
- id: evo
name: "Evo"
type: model
url: https://github.com/evo-design/evo
description: "Long-context genomic foundation model (up to 1M tokens)."
tags: [foundation-models, genomics-foundation-models]
tasks: [Foundation Model]
modalities: [Genomics]
organism: []
api: false
- id: evodiff
name: "EvoDiff"
type: model
url: https://github.com/microsoft/evodiff
description: "Discrete diffusion framework for protein sequence generation trained on evolutionary-scale data, supporting unconditional generation, disordered region design, and functional motif scaffolding. [ [paper-2023](https://www.biorxiv.org/content/10.1101/2023.09.11.556673v1) ]"
tags: [foundation-models, protein-foundation-models, protein-structure-prediction-and-design]
tasks: [Foundation Model, Protein Structure Prediction]
modalities: [Protein]
organism: []
api: false
- id: evolutionary_scale_modeling_esm
name: "Evolutionary Scale Modeling (ESM)"
type: model
url: https://github.com/facebookresearch/esm
description: "Protein embeddings."
tags: [foundation-models, pre-trained-embedding, protein-foundation-models]
tasks: [Foundation Model]
modalities: [Protein]
organism: []
api: false
- id: gears
name: "GEARS"
type: model
url: https://github.com/snap-stanford/GEARS
description: "Graph-based model for predicting transcriptional responses to single and combinatorial genetic perturbations using biological priors."
tags: [foundation-models, single-cell-foundation-models, transcriptomics-foundation-models]
tasks: [Foundation Model]
modalities: [Single Cell, Transcriptomics]
organism: []
api: false
- id: genecompass
name: "GeneCompass"
type: model
url: https://github.com/xCompass-AI/GeneCompass
description: "Large-scale foundation model integrating DNA regulatory sequences and single-cell transcriptomics from 120M+ cells across multiple species for gene regulation prediction."
tags: [foundation-models, multi-omics-foundation-models, single-cell-foundation-models]
tasks: [Foundation Model]
modalities: [Multi-Omics, Single Cell]
organism: []
api: false
- id: geneformer
name: "Geneformer"
type: model
url: https://huggingface.co/ctheodoris/Geneformer
description: "Context-aware, attention-based deep learning model pretrained on a large corpus of single-cell transcriptomes."
tags: [foundation-models, single-cell-foundation-models, transcriptomics-foundation-models]
tasks: [Foundation Model]
modalities: [Single Cell, Transcriptomics]
organism: []
api: false
- id: genegpt
name: "GeneGPT"
type: model
url: https://github.com/ncbi/GeneGPT
description: "LLM for biomedical information, integrated with various APIs."
tags: [llm-for-biology]
tasks: [Language Modeling]
modalities: [Text]
organism: []
api: false
- id: genept
name: "GenePT"
type: model
url: https://github.com/yiqunchen/GenePT
description: "Foundation LLM for single-cell data."
tags: [llm-for-biology]
tasks: [Language Modeling]
modalities: [Text]
organism: []
api: false
- id: gigapath
name: "GigaPath"
type: model
url: https://github.com/prov-gigapath/prov-gigapath
description: "Slide-level digital pathology foundation model pretrained on 1.3 billion pathology image tokens from whole-slide images."
tags: [foundation-models, single-cell-foundation-models, spatial-foundation-models]
tasks: [Foundation Model]
modalities: [Single Cell, Spatial Transcriptomics]
organism: []
api: false
- id: glue
name: "GLUE"
type: model
url: https://github.com/gao-lab/GLUE
description: "Graph-Linked Unified Embedding framework for unpaired single-cell multi-omics data integration across RNA, ATAC, methylation, and protein modalities."
tags: [foundation-models, multi-omics-foundation-models, single-cell-foundation-models]
tasks: [Foundation Model]
modalities: [Multi-Omics, Single Cell]
organism: []
api: false
- id: gpn_genomic_pre_trained_network
name: "GPN (Genomic Pre-trained Network)"
type: model
url: https://github.com/songlab-cal/gpn
description: "Masked language model for DNA sequences enabling zero-shot variant effect prediction without requiring functional annotations."
tags: [foundation-models, genomics-foundation-models]
tasks: [Foundation Model]
modalities: [Genomics]
organism: []
api: false
- id: graphdta
name: "GraphDTA"
type: model
url: https://github.com/thinng/GraphDTA
description: "Graph neural networkโ€“based DTI prediction using molecular graphs."
tags: [drug-discovery, drug-target-interaction]
tasks: [Drug Discovery, Drug Target Interaction]
modalities: [Protein, Small Molecule]
organism: []
api: false
- id: grover
name: "GROVER"
type: model
url: https://github.com/tencent-ailab/grover
description: "Self-supervised graph transformer for large-scale molecular representation learning from unlabeled compounds."
tags: [compound-embedding, compound-foundation-models, foundation-models]
tasks: [Foundation Model]
modalities: [Small Molecule]
organism: []
api: false
- id: hidra
name: "HiDRA"
type: model
url: https://github.com/bsml320/HiDRA
description: "Hierarchical network model incorporating gene and pathway-level information for cancer drug response prediction."
tags: [drug-discovery, drug-response-prediction]
tasks: [Drug Discovery, Drug Response Prediction]
modalities: [Small Molecule]
organism: []
api: false
- id: hyenadna
name: "HyenaDNA"
type: model
url: https://github.com/HazyResearch/hyena-dna
description: "Long-range genomic foundation model handling sequences up to 1M tokens with sub-quadratic attention."
tags: [foundation-models, genomics-foundation-models]
tasks: [Foundation Model]
modalities: [Genomics]
organism: []
api: false
- id: jamie
name: "JAMIE"
type: model
url: https://github.com/Oafish1/JAMIE
description: "Joint variational autoencoder for multimodal single-cell data imputation and embedding."
tags: [foundation-models, multi-omics-foundation-models, single-cell-foundation-models]
tasks: [Foundation Model]
modalities: [Multi-Omics, Single Cell]
organism: []
api: false
- id: jtvae
name: "JTVAE"
type: model
url: https://github.com/wengong-jin/icml18-jtnn
description: "Junction tree variational autoencoder for molecular graph generation that guarantees chemical validity via a hierarchical tree decomposition."
tags: [drug-discovery, molecular-generation]
tasks: [Drug Discovery, Molecular Generation]
modalities: [Small Molecule]
organism: []
api: false
- id: matcha
name: "Matcha"
type: model
url: https://github.com/LigandPro/Matcha
description: "Multi-stage Riemannian flow matching model for physically valid molecular docking with scoring, pose filtering, and benchmarks."
tags: [drug-discovery, molecular-generation]
tasks: [Drug Discovery, Molecular Generation]
modalities: [Small Molecule]
organism: []
api: false
- id: mcpinn
name: "MCPINN"
type: model
url: https://github.com/mhlee0903/multi_channels_PINN
description: "Drug discovery via compound-protein interaction and machine learning."
tags: [compound-protein-interaction, drug-discovery]
tasks: [Compound-Protein Interaction, Drug Discovery]
modalities: [Protein, Small Molecule]
organism: []
api: false
- id: midas
name: "MIDAS"
type: model
url: https://github.com/labomics/midas
description: "Mosaic integration and differential accessibility model for single-cell multi-omics that handles arbitrary missing-modality combinations across transcriptomics, chromatin accessibility, and proteomics."
tags: [foundation-models, multi-omics-foundation-models, single-cell-foundation-models]
tasks: [Foundation Model]
modalities: [Multi-Omics, Single Cell]
organism: []
api: false
- id: mira
name: "MIRA"
type: model
url: https://github.com/cistrome/MIRA
description: "Probabilistic multimodal topic model jointly modeling single-cell transcriptomics and chromatin accessibility for regulatory network inference."
tags: [foundation-models, multi-omics-foundation-models, single-cell-foundation-models]
tasks: [Foundation Model]
modalities: [Multi-Omics, Single Cell]
organism: []
api: false
- id: mofa
name: "MOFA+"
type: model
url: https://github.com/bioFAM/MOFA2
description: "Multi-Omics Factor Analysis framework identifying shared axes of variation across bulk and single-cell datasets including RNA, ATAC, proteomics, methylation, and copy number."
tags: [foundation-models, multi-omics-foundation-models, single-cell-foundation-models]
tasks: [Foundation Model]
modalities: [Multi-Omics, Single Cell]
organism: []
api: false
- id: mofgcn
name: "MOFGCN"
type: model
url: https://github.com/weiba/MOFGCN/tree/main
description: "GCN + heterogeneous network."
tags: [drug-discovery, drug-response-prediction]
tasks: [Drug Discovery, Drug Response Prediction]
modalities: [Small Molecule]
organism: []
api: false
- id: mol2vec
name: "Mol2Vec"
type: model
url: https://github.com/samoturk/mol2vec
description: "Unsupervised molecular embedding method inspired by Word2Vec for learning vector representations of chemical substructures."
tags: [compound-embedding, compound-foundation-models, foundation-models]
tasks: [Foundation Model]
modalities: [Small Molecule]
organism: []
api: false
- id: molecular_transformer
name: "Molecular Transformer"
type: model
url: https://github.com/pschwllr/MolecularTransformer
description: "Sequence-to-sequence model for retrosynthesis prediction."
tags: [drug-discovery, molecular-generation]
tasks: [Drug Discovery, Molecular Generation]
modalities: [Small Molecule]
organism: []
api: false
- id: molformer
name: "MolFormer"
type: model
url: https://github.com/IBM/molformer
description: "Linear attention transformer pretrained on millions of SMILES strings for efficient molecular embeddings."
tags: [compound-embedding, compound-foundation-models, foundation-models]
tasks: [Foundation Model]
modalities: [Small Molecule]
organism: []
api: false
- id: molgpt
name: "MolGPT"
type: model
url: https://github.com/devalab/molgpt
description: "Transformer-based model for molecular generation."
tags: [drug-discovery, molecular-generation]
tasks: [Drug Discovery, Molecular Generation]
modalities: [Small Molecule]
organism: []
api: false
- id: molt5
name: "MolT5"
type: model
url: https://github.com/blender-nlp/MolT5
description: "Language model for molecular tasks bridging text and SMILES, enabling molecule captioning and text-driven molecule generation."
tags: [llm-for-biology]
tasks: [Language Modeling]
modalities: [Text]
organism: []
api: false
- id: moltrans
name: "MolTrans"
type: model
url: https://github.com/kexinhuang12345/MolTrans
description: "Transformer-based DTI model leveraging molecular substructures."
tags: [drug-discovery, drug-target-interaction]
tasks: [Drug Discovery, Drug Target Interaction]
modalities: [Protein, Small Molecule]
organism: []
api: false
- id: multigrate
name: "Multigrate"
type: model
url: https://github.com/theislab/multigrate
description: "Asymmetric multi-omics variational autoencoder for integrating single-cell data across RNA, ATAC, and protein modalities with missing-modality support."
tags: [foundation-models, multi-omics-foundation-models, single-cell-foundation-models]
tasks: [Foundation Model]
modalities: [Multi-Omics, Single Cell]
organism: []
api: false
- id: multivi
name: "MultiVI"
type: model
url: https://github.com/scverse/scvi-tools
description: "Multi-modal variational autoencoder for integrating paired and unpaired single-cell RNA-seq and ATAC-seq measurements into a unified latent space."
tags: [foundation-models, multi-omics-foundation-models, single-cell-foundation-models]
tasks: [Foundation Model]
modalities: [Multi-Omics, Single Cell]
organism: []
api: false
- id: musk
name: "MUSK"
type: model
url: https://github.com/lilab-stanford/MUSK
description: "Vision-language foundation model for precision oncology analyzing multimodal paired text and pathology image data for biomarker prediction and retrieval."
tags: [foundation-models, multi-modal-foundation-models]
tasks: [Foundation Model]
modalities: [Multi-Modal]
organism: []
api: false
- id: nbbayeslm
name: "NbBayesLM"
type: model
url: https://github.com/FairuzShadmaniShishir/NbBayesLM
description: "Bayesian neural network integrating protein language model embeddings and physicochemical features to predict nanobody thermostability with uncertainty estimates. [Paper](https://www.frontiersin.org/journals/bioinformatics/articles/10.3389/fbinf.2026.1832968/full)"
tags: [protein-property-prediction]
tasks: [Protein Property Prediction]
modalities: [Protein]
organism: []
api: false
- id: neodti
name: "NeoDTI"
type: model
url: https://github.com/FangpingWan/NeoDTI
description: "Library for drug-target interaction prediction."
tags: [drug-discovery, drug-target-interaction]
tasks: [Drug Discovery, Drug Target Interaction]
modalities: [Protein, Small Molecule]
organism: []
api: false
- id: nicheformer
name: "Nicheformer"
type: model
url: https://github.com/theislab/nicheformer
description: "Foundation model for single-cell and spatial omics using a transformer architecture with positional embeddings to encode spatial cell information."
tags: [foundation-models, single-cell-foundation-models, spatial-foundation-models]
tasks: [Foundation Model]
modalities: [Single Cell, Spatial Transcriptomics]
organism: []
api: false
- id: nucleotide_transformer
name: "Nucleotide Transformer"
type: model
url: https://github.com/instadeepai/nucleotide-transformer
description: "Foundation model for genomic sequences across multiple species."
tags: [foundation-models, genomics-foundation-models]
tasks: [Foundation Model]
modalities: [Genomics]
organism: []
api: false
- id: omegafold
name: "OmegaFold"
type: model
url: https://github.com/HeliXonProtein/OmegaFold
description: "High-resolution de novo protein structure prediction from sequence."
tags: [foundation-models, protein-foundation-models, protein-structure-prediction-and-design]
tasks: [Foundation Model, Protein Structure Prediction]
modalities: [Protein]
organism: []
api: false
- id: openfold
name: "OpenFold"
type: model
url: https://github.com/aqlaboratory/openfold
description: "Trainable, memory-efficient open-source reproduction of AlphaFold2 enabling custom protein structure prediction workflows."
tags: [foundation-models, protein-foundation-models, protein-structure-prediction-and-design]
tasks: [Foundation Model, Protein Structure Prediction]
modalities: [Protein]
organism: []
api: false
- id: paccmannrl
name: "PaccMannRL"
type: model
url: https://github.com/PaccMann/paccmann_generator
description: "Reinforcement learning-based generative model for de novo hit-like anticancer molecule design from transcriptomic data."
tags: [drug-discovery, molecular-generation]
tasks: [Drug Discovery, Molecular Generation]
modalities: [Small Molecule]
organism: []
api: false
- id: pathomicfusion
name: "PathomicFusion"
type: model
url: https://github.com/mahmoodlab/PathomicFusion
description: "Integrated framework fusing histopathology and genomic features via CNN, GNN, and attention gating for cancer diagnosis and prognosis."
tags: [foundation-models, multi-modal-foundation-models]
tasks: [Foundation Model]
modalities: [Multi-Modal]
organism: []
api: false
- id: phikon
name: "Phikon"
type: model
url: https://huggingface.co/owkin/phikon
description: "ViT-based pathology foundation model pretrained with iBOT self-supervision on TCGA whole-slide images."
tags: [foundation-models, single-cell-foundation-models, spatial-foundation-models]
tasks: [Foundation Model]
modalities: [Single Cell, Spatial Transcriptomics]
organism: []
api: false
- id: plip
name: "PLIP"
type: model
url: https://github.com/PathologyFoundation/plip
description: "Vision-language foundation model for pathology trained with contrastive learning on pathology imageโ€“text pairs for image classification and text-to-image retrieval."
tags: [foundation-models, multi-modal-foundation-models]
tasks: [Foundation Model]
modalities: [Multi-Modal]
organism: []
api: false
- id: porpoise
name: "PORPOISE"
type: model
url: https://github.com/mahmoodlab/PORPOISE
description: "Pan-cancer integrative histology-genomic analysis framework using multimodal deep learning for patient stratification."
tags: [foundation-models, multi-modal-foundation-models]
tasks: [Foundation Model]
modalities: [Multi-Modal]
organism: []
api: false
- id: prnet
name: "PRNet"
type: model
url: https://github.com/Perturbation-Response-Prediction/PRnet
description: "Deep generative model for predicting transcriptional responses to novel chemical perturbations for drug discovery."
tags: [drug-discovery, drug-perturbation]
tasks: [Drug Discovery, Drug Perturbation]
modalities: [Small Molecule]
organism: []
api: false
- id: progen2
name: "ProGen2"
type: model
url: https://github.com/salesforce/progen
description: "Protein language model trained on diverse protein families for sequence generation and fitness prediction."
tags: [foundation-models, pre-trained-embedding, protein-foundation-models]
tasks: [Foundation Model]
modalities: [Protein]
organism: []
api: false
- id: proteinmpnn
name: "ProteinMPNN"
type: model
url: https://github.com/dauparas/ProteinMPNN
description: "Deep learning model for protein sequence design given backbone structure."
tags: [foundation-models, protein-foundation-models, protein-structure-prediction-and-design]
tasks: [Foundation Model, Protein Structure Prediction]
modalities: [Protein]
organism: []
api: false
- id: prottrans
name: "ProtTrans"
type: model
url: https://github.com/agemagician/ProtTrans
description: "Suite of protein language models (ProtBERT, ProtT5, ProtXLNet) trained on billions of protein sequences from UniRef and BFD."
tags: [foundation-models, pre-trained-embedding, protein-foundation-models]
tasks: [Foundation Model]
modalities: [Protein]
organism: []
api: false
- id: recover
name: "RECOVER"
type: model
url: https://github.com/RECOVERcoalition/Recover
description: "Machine learning framework for predicting synergistic drug combination responses across cell lines."
tags: [drug-discovery, drug-response-prediction]
tasks: [Drug Discovery, Drug Response Prediction]
modalities: [Small Molecule]
organism: []
api: false
- id: reinvent
name: "REINVENT"
type: model
url: https://github.com/MolecularAI/Reinvent
description: "Reinforcement learning for de novo drug design."
tags: [drug-discovery, molecular-generation]
tasks: [Drug Discovery, Molecular Generation]
modalities: [Small Molecule]
organism: []
api: false
- id: release
name: "ReLeaSE"
type: model
url: https://github.com/isayev/ReLeaSE
description: "Deep reinforcement learning framework for de novo drug design combining a generative and predictive model."
tags: [drug-discovery, molecular-generation]
tasks: [Drug Discovery, Molecular Generation]
modalities: [Small Molecule]
organism: []
api: false
- id: rfdiffusion
name: "RFdiffusion"
type: model
url: https://github.com/RosettaCommons/RFdiffusion
description: "Generative model for protein backbone design using diffusion."
tags: [foundation-models, protein-foundation-models, protein-structure-prediction-and-design]
tasks: [Foundation Model, Protein Structure Prediction]
modalities: [Protein]
organism: []
api: false
- id: rosettafold
name: "RoseTTAFold"
type: model
url: https://github.com/RosettaCommons/RoseTTAFold
description: "Three-track neural network for protein structure prediction."
tags: [foundation-models, protein-foundation-models, protein-structure-prediction-and-design]
tasks: [Foundation Model, Protein Structure Prediction]
modalities: [Protein]
organism: []
api: false
- id: saprot
name: "SaProt"
type: model
url: https://github.com/westlake-reup/SaProt
description: "Structure-aware protein language model using structure-aware tokens that encode both sequence and backbone geometry for improved function prediction."
tags: [foundation-models, protein-foundation-models, protein-structure-prediction-and-design]
tasks: [Foundation Model, Protein Structure Prediction]
modalities: [Protein]
organism: []
api: false
- id: saturn
name: "SATURN"
type: model
url: https://github.com/snap-stanford/SATURN
description: "Transformer-based model integrating gene expression and protein sequences via a protein language model to learn unified multi-species cell embeddings."
tags: [foundation-models, single-cell-foundation-models, transcriptomics-foundation-models]
tasks: [Foundation Model]
modalities: [Single Cell, Transcriptomics]
organism: []
api: false
- id: scarches
name: "scArches"
type: model
url: https://github.com/theislab/scarches
description: "Transfer learning framework for mapping new single-cell datasets onto pre-trained reference atlases across batches, conditions, and modalities."
tags: [domain-alignment, foundation-models, single-cell-foundation-models]
tasks: [Domain Alignment, Foundation Model]
modalities: [Single Cell]
organism: []
api: false
- id: scbert
name: "scBERT"
type: model
url: https://github.com/TencentAILabHealthcare/scBERT
description: "BERT-based foundation model pretrained on large-scale scRNA-seq data for cell type annotation."
tags: [foundation-models, single-cell-foundation-models, transcriptomics-foundation-models]
tasks: [Foundation Model]
modalities: [Single Cell, Transcriptomics]
organism: []
api: false
- id: scbutterfly
name: "scButterfly"
type: model
url: https://github.com/BioX-NKU/scButterfly
description: "Dual-aligned variational autoencoder for single-cell cross-modality translation between paired and unpaired multiomics data."
tags: [foundation-models, multi-omics-foundation-models, single-cell-foundation-models]
tasks: [Foundation Model]
modalities: [Multi-Omics, Single Cell]
organism: []
api: false
- id: scfoundation
name: "scFoundation"
type: model
url: https://github.com/biomap-research/scFoundation
description: "Large-scale foundation model for single-cell gene expression, enabling multiple downstream tasks."
tags: [foundation-models, single-cell-foundation-models, transcriptomics-foundation-models]
tasks: [Foundation Model]
modalities: [Single Cell, Transcriptomics]
organism: []
api: false
- id: scgpt
name: "scGPT"
type: model
url: https://github.com/bowang-lab/scGPT
description: "Transformer-based foundation model pretrained on millions of single-cell profiles."
tags: [foundation-models, single-cell-foundation-models, transcriptomics-foundation-models]
tasks: [Foundation Model]
modalities: [Single Cell, Transcriptomics]
organism: []
api: false
- id: scgpt_spatial
name: "scGPT-spatial"
type: model
url: https://github.com/bowang-lab/scGPT-spatial
description: "Extension of scGPT for spatial transcriptomics with continual pretraining and a mixture-of-experts decoder for spatial gene expression analysis."
tags: [foundation-models, single-cell-foundation-models, spatial-foundation-models]
tasks: [Foundation Model]
modalities: [Single Cell, Spatial Transcriptomics]
organism: []
api: false
- id: scmulan
name: "scMulan"
type: model
url: https://github.com/SuperBianC/scMulan
description: "Single-cell multi-omic language model pretrained on ~10M cells spanning transcriptomics, epigenomics, and proteomics for cross-omics transfer tasks."
tags: [foundation-models, multi-omics-foundation-models, single-cell-foundation-models]
tasks: [Foundation Model]
modalities: [Multi-Omics, Single Cell]
organism: []
api: false
- id: scpair
name: "scPair"
type: model
url: https://github.com/quon-titative-biology/scPair
description: "Bidirectional feedforward network for single-cell multimodal analysis with cross-modality prediction leveraging single-cell atlases."
tags: [foundation-models, multi-omics-foundation-models, single-cell-foundation-models]
tasks: [Foundation Model]
modalities: [Multi-Omics, Single Cell]
organism: []
api: false
- id: scprint
name: "scPRINT"
type: model
url: https://github.com/cantinilab/scPRINT
description: "Pretrained on 50M cells for scRNA-seq denoising & zero imputation."
tags: [llm-for-biology]
tasks: [Language Modeling]
modalities: [Text]
organism: []
api: false
- id: sei
name: "Sei"
type: model
url: https://github.com/FunctionLab/sei-framework
description: "Sequence-to-function framework learning a genome-wide regulatory activity code from DNA sequences for variant effect prediction."
tags: [foundation-models, genomics-foundation-models]
tasks: [Foundation Model]
modalities: [Genomics]
organism: []
api: false
- id: spatialglue
name: "SpatialGlue"
type: model
url: https://github.com/zhanglabtools/SpatialGlue
description: "Graph attention network for spatial multi-omics integration jointly embedding spatial transcriptomics with chromatin accessibility or proteomics."
tags: [foundation-models, multi-omics-foundation-models, single-cell-foundation-models]
tasks: [Foundation Model]
modalities: [Multi-Omics, Single Cell]
organism: []
api: false
- id: targetdiff
name: "TargetDiff"
type: model
url: https://github.com/guanjq/targetdiff
description: "3D equivariant diffusion model for structure-based drug design."
tags: [drug-discovery, molecular-generation]
tasks: [Drug Discovery, Molecular Generation]
modalities: [Small Molecule]
organism: []
api: false
- id: tgsa
name: "TGSA"
type: model
url: https://github.com/violet-sto/TGSA
description: "Tumor gene set and attention-based model leveraging biological pathway knowledge for drug response prediction."
tags: [drug-discovery, drug-response-prediction]
tasks: [Drug Discovery, Drug Response Prediction]
modalities: [Small Molecule]
organism: []
api: false
- id: toad
name: "TOAD"
type: model
url: https://github.com/mahmoodlab/TOAD
description: "Tumor Origin Assessment via Deep-learning; weakly-supervised multi-task model predicting cancer primary origin from H&E whole-slide images."
tags: [foundation-models, multi-modal-foundation-models]
tasks: [Foundation Model]
modalities: [Multi-Modal]
organism: []
api: false
- id: tosica
name: "TOSICA"
type: model
url: https://github.com/JackieHanlaopo/TOSICA
description: "Transformer-based framework for one-stop interpretable cell-type annotation supporting cross-dataset and cross-species transfer."
tags: [domain-alignment, foundation-models, single-cell-foundation-models]
tasks: [Domain Alignment, Foundation Model]
modalities: [Single Cell]
organism: []
api: false
- id: totalvi
name: "totalVI"
type: model
url: https://github.com/scverse/scvi-tools
description: "Probabilistic framework for joint analysis of paired scRNA-seq and protein (CITE-seq) data enabling multi-modal cell state representation across single-cell datasets."
tags: [foundation-models, multi-omics-foundation-models, single-cell-foundation-models]
tasks: [Foundation Model]
modalities: [Multi-Omics, Single Cell]
organism: []
api: false
- id: transformercpi
name: "TransformerCPI"
type: model
url: https://github.com/lifanchen-simm/transformerCPI
description: "CPI prediction using Transformer."
tags: [compound-protein-interaction, drug-discovery]
tasks: [Compound-Protein Interaction, Drug Discovery]
modalities: [Protein, Small Molecule]
organism: []
api: false
- id: transigen
name: "TranSiGen"
type: model
url: https://github.com/myzhengSIMM/TranSiGen
description: "Dual-VAE architecture for ligand-based virtual screening, drug response prediction, and drug repurposing using chemical-induced transcriptional profiles."
tags: [drug-discovery, drug-repurposing]
tasks: [Drug Discovery, Drug Repurposing]
modalities: [Small Molecule]
organism: []
api: false
- id: uce
name: "UCE"
type: model
url: https://github.com/snap-stanford/UCE
description: "Universal Cell Embeddings: zero-shot single-cell embedding model trained on 36M cells across species, tissues, and assays without fine-tuning."
tags: [foundation-models, single-cell-foundation-models, transcriptomics-foundation-models]
tasks: [Foundation Model]
modalities: [Single Cell, Transcriptomics]
organism: []
api: false
- id: uni
name: "UNI"
type: model
url: https://github.com/mahmoodlab/UNI
description: "General-purpose self-supervised pathology foundation model trained on 100K+ whole-slide images for diverse computational pathology tasks."
tags: [foundation-models, single-cell-foundation-models, spatial-foundation-models]
tasks: [Foundation Model]
modalities: [Single Cell, Spatial Transcriptomics]
organism: []
api: false
- id: uni_mol
name: "Uni-Mol"
type: model
url: https://github.com/deepmodeling/Uni-Mol
description: "3D molecular pretraining framework for universal representation learning on molecules and protein pockets."
tags: [compound-embedding, compound-foundation-models, foundation-models]
tasks: [Foundation Model]
modalities: [Small Molecule]
organism: []
api: false
- id: unitednet
name: "UnitedNet"
type: model
url: https://github.com/LiuLab-Bioelectronics-Harvard/UnitedNet
description: "Interpretable multi-task deep neural network for single-cell multi-omics integration spanning transcriptomics, chromatin accessibility, and proteomics."
tags: [foundation-models, multi-omics-foundation-models, single-cell-foundation-models]
tasks: [Foundation Model]
modalities: [Multi-Omics, Single Cell]
organism: []
api: false
- id: virchow
name: "Virchow"
type: model
url: https://huggingface.co/paige-ai/Virchow
description: "Million-slide digital pathology foundation model using a vision transformer and self-supervised distillation for tile-level pathology image representation."
tags: [foundation-models, multi-modal-foundation-models]
tasks: [Foundation Model]
modalities: [Multi-Modal]
organism: []
api: false
- id: autozyme
name: "AutoZyme"
type: toolkit
url: https://github.com/ElliotXie/autozyme
description: "Autonomous agentic framework that speeds up bioinformatics software (e.g. Scanpy, Seurat) on CPUs while preserving the original results."
tags: [preprocessing-tools]
tasks: [Preprocessing]
modalities: []
organism: []
api: false
- id: biopython
name: "Biopython"
type: toolkit
url: https://biopython.org/
description: "Collection of Python tools for biological computation including sequence analysis, structure parsing, and database access."
tags: [preprocessing-tools]
tasks: [Preprocessing]
modalities: []
organism: []
api: false
- id: casper
name: "CaSpER"
type: toolkit
url: https://github.com/akdess/CaSpER
description: "CNV identification and visualization by integrative analysis of single-cell or bulk RNA-seq data."
tags: [preprocessing-tools]
tasks: [Preprocessing]
modalities: []
organism: []
api: false
- id: cellcharter
name: "CellCharter"
type: toolkit
url: https://github.com/CSOgroup/cellcharter
description: "Identification and characterization of spatial cell niches from spatial transcriptomics using VAEs and Gaussian mixture models."
tags: [preprocessing-tools]
tasks: [Preprocessing]
modalities: []
organism: []
api: false
- id: cellchat
name: "CellChat"
type: toolkit
url: https://github.com/sqjin/CellChat
description: "Inference and analysis of cell-cell communication ligand-receptor networks from single-cell transcriptomics data."
tags: [preprocessing-tools]
tasks: [Preprocessing]
modalities: []
organism: []
api: false
- id: celltypist
name: "CellTypist"
type: toolkit
url: https://github.com/Teichlab/celltypist
description: "Automated cell type annotation for scRNA-seq."
tags: [preprocessing-tools]
tasks: [Preprocessing]
modalities: []
organism: []
api: false
- id: chatspatial
name: "ChatSpatial"
type: toolkit
url: https://github.com/cafferychen777/ChatSpatial
description: "MCP server for spatial transcriptomics analysis via natural language."
tags: [preprocessing-tools]
tasks: [Preprocessing]
modalities: []
organism: []
api: false
- id: chemistry_development_kit
name: "Chemistry Development Kit"
type: toolkit
url: https://github.com/cdk/cdk
description: "Cheminformatics software & machine learning tools."
tags: [preprocessing-tools]
tasks: [Preprocessing]
modalities: []
organism: []
api: false
- id: commot
name: "COMMOT"
type: toolkit
url: https://github.com/zcang/COMMOT
description: "Optimal transport-based framework for screening cell-cell communication in spatial transcriptomics."
tags: [preprocessing-tools]
tasks: [Preprocessing]
modalities: []
organism: []
api: false
- id: deepchem
name: "DeepChem"
type: toolkit
url: https://github.com/deepchem/deepchem
description: "Deep learning library for drug discovery, quantum chemistry, and materials science."
tags: [preprocessing-tools]
tasks: [Preprocessing]
modalities: []
organism: []
api: false
- id: deeptalk
name: "DeepTalk"
type: toolkit
url: https://github.com/JiangBioLab/DeepTalk
description: "Graph attention network for deciphering cell-cell communication from spatial transcriptomics."
tags: [preprocessing-tools]
tasks: [Preprocessing]
modalities: []
organism: []
api: false
- id: doubletfinder
name: "DoubletFinder"
type: toolkit
url: https://github.com/chris-mcginnis-ucsf/DoubletFinder
description: "Machine learning approach for detecting multiplet (doublet) artifacts in single-cell RNA-seq data."
tags: [preprocessing-tools]
tasks: [Preprocessing]
modalities: []
organism: []
api: false
- id: flashdeconv
name: "FlashDeconv"
type: toolkit
url: https://github.com/cafferychen777/flashdeconv
description: "High-performance spatial transcriptomics deconvolution (~1M spots in ~3 min)."
tags: [preprocessing-tools]
tasks: [Preprocessing]
modalities: []
organism: []
api: false
- id: gromacs
name: "GROMACS"
type: toolkit
url: https://www.gromacs.org/
description: "Molecular dynamics simulation package for biochemical molecules."
tags: [preprocessing-tools]
tasks: [Preprocessing]
modalities: []
organism: []
api: false
- id: harmony
name: "Harmony"
type: toolkit
url: https://github.com/immunogenomics/harmony
description: "Fast and scalable integration of single-cell data across datasets, conditions, technologies, and species."
tags: [preprocessing-tools]
tasks: [Preprocessing]
modalities: []
organism: []
api: false
- id: kallisto
name: "kallisto"
type: toolkit
url: https://pachterlab.github.io/kallisto/
description: "Near-optimal RNA-seq quantification using pseudoalignment for fast transcript abundance estimation."
tags: [preprocessing-tools]
tasks: [Preprocessing]
modalities: []
organism: []
api: false
- id: linger
name: "LINGER"
type: toolkit
url: https://github.com/Durenlab/LINGER
description: "Neural network for gene regulatory network inference from single-cell multiome (RNA+ATAC-seq) data with bulk data pretraining."
tags: [preprocessing-tools]
tasks: [Preprocessing]
modalities: []
organism: []
api: false
- id: mdanalysis
name: "MDAnalysis"
type: toolkit
url: https://www.mdanalysis.org/
description: "Python library for analyzing and altering molecular dynamics simulation trajectories."
tags: [preprocessing-tools]
tasks: [Preprocessing]
modalities: []
organism: []
api: false
- id: mogonet
name: "MOGONET"
type: toolkit
url: https://github.com/txWang/MOGONET
description: "Multi-omics graph convolutional network framework for patient classification and biomarker identification."
tags: [preprocessing-tools]
tasks: [Preprocessing]
modalities: []
organism: []
api: false
- id: monocle3
name: "Monocle3"
type: toolkit
url: https://cole-trapnell-lab.github.io/monocle3/
description: "Single-cell trajectory analysis tool for learning developmental trajectories and ordering cells in pseudotime."
tags: [preprocessing-tools]
tasks: [Preprocessing]
modalities: []
organism: []
api: false
- id: ncem
name: "NCEM"
type: toolkit
url: https://github.com/theislab/ncem
description: "GNN-based model for learning intercellular communication from spatial graphs of cells."
tags: [preprocessing-tools]
tasks: [Preprocessing]
modalities: []
organism: []
api: false
- id: numbat
name: "Numbat"
type: toolkit
url: https://github.com/kharchenkolab/numbat
description: "Haplotype-aware copy number variation inference from single-cell RNA-seq using hidden Markov models."
tags: [preprocessing-tools]
tasks: [Preprocessing]
modalities: []
organism: []
api: false
- id: openmm
name: "OpenMM"
type: toolkit
url: https://openmm.org/
description: "High-performance toolkit for molecular simulation and GPU-accelerated MD."
tags: [preprocessing-tools]
tasks: [Preprocessing]
modalities: []
organism: []
api: false
- id: rdkit
name: "RDKit"
type: toolkit
url: https://github.com/rdkit/rdkit
description: "Cheminformatics software & machine learning toolkit."
tags: [preprocessing-tools]
tasks: [Preprocessing]
modalities: []
organism: []
api: false
- id: scanpy
name: "Scanpy"
type: toolkit
url: https://scanpy.readthedocs.io/en/stable/
description: "Python library for scRNA-seq analysis."
tags: [preprocessing-tools]
tasks: [Preprocessing]
modalities: []
organism: []
api: false
- id: scenic
name: "SCENIC"
type: toolkit
url: https://github.com/aertslab/SCENIC
description: "Single-cell regulatory network inference and clustering linking transcription factors to co-expressed gene modules."
tags: [preprocessing-tools]
tasks: [Preprocessing]
modalities: []
organism: []
api: false
- id: scipenn
name: "sciPENN"
type: toolkit
url: https://github.com/jlakkis/sciPENN
description: "RNN-based method for simultaneous protein expression prediction, uncertainty estimation, and cell-type label transfer from CITE-seq and scRNA-seq data."
tags: [preprocessing-tools]
tasks: [Preprocessing]
modalities: []
organism: []
api: false
- id: scvelo
name: "scVelo"
type: toolkit
url: https://github.com/theislab/scvelo
description: "RNA velocity estimation for single-cell transcriptomics, inferring the direction and speed of cell differentiation."
tags: [preprocessing-tools]
tasks: [Preprocessing]
modalities: []
organism: []
api: false
- id: scvi_tools
name: "scvi-tools"
type: toolkit
url: https://scvi-tools.org/
description: "Probabilistic models for single-cell omics data analysis."
tags: [preprocessing-tools]
tasks: [Preprocessing]
modalities: []
organism: []
api: false
- id: seqbench
name: "SeqBench"
type: toolkit
url: https://seqbench.com/
description: "Web-based molecular biology sequence workbench for primer design, cloning simulation (Gibson, Golden Gate, restriction digest), CRISPR guide RNA design, and sequence analysis, with a public REST API, OpenAPI 3.1 spec, and MCP server."
tags: [preprocessing-tools]
tasks: [Preprocessing]
modalities: []
organism: []
api: false
- id: seurat
name: "Seurat"
type: toolkit
url: https://satijalab.org/seurat/
description: "R library for scRNA-seq analysis."
tags: [preprocessing-tools]
tasks: [Preprocessing]
modalities: []
organism: []
api: false
- id: squidpy
name: "Squidpy"
type: toolkit
url: https://squidpy.readthedocs.io/
description: "Python library for spatial single-cell analysis."
tags: [preprocessing-tools]
tasks: [Preprocessing]
modalities: []
organism: []
api: false
- id: stagate
name: "STAGATE"
type: toolkit
url: https://github.com/RucDongLab/STAGATE
description: "Adaptive graph attention auto-encoder for spatial domain identification in spatial transcriptomics."
tags: [preprocessing-tools]
tasks: [Preprocessing]
modalities: []
organism: []
api: false
- id: star
name: "STAR"
type: toolkit
url: https://github.com/alexdobin/STAR
description: "Ultrafast universal RNA-seq aligner with support for spliced alignment and single-cell quantification via STARsolo."
tags: [preprocessing-tools]
tasks: [Preprocessing]
modalities: []
organism: []
api: false
- id: tigon
name: "TIGON"
type: toolkit
url: https://github.com/yutongo/TIGON
description: "Neural optimal transport method for reconstructing growth and dynamic trajectories from single-cell transcriptomics."
tags: [preprocessing-tools]
tasks: [Preprocessing]
modalities: []
organism: []
api: false