5980 lines
147 KiB
JSON
5980 lines
147 KiB
JSON
---
|
||
title: "Resources"
|
||
task: ""
|
||
lineage_type: import
|
||
upstream_source: https://github.com/inoue0426/awesome-computational-biology/blob/7a064bf0/data/resources.json
|
||
upstream_sha: 7a064bf0
|
||
imported_at: 2026-08-08
|
||
prompt_class: catalogue
|
||
upstream_changes: accepted
|
||
author: upstream
|
||
validated: false
|
||
---
|
||
|
||
[
|
||
{
|
||
"id": "chembl_web_services",
|
||
"name": "ChEMBL Web Services",
|
||
"type": "api",
|
||
"url": "https://www.ebi.ac.uk/chembl/ws",
|
||
"description": "REST API for bioactive molecules, targets, and bioassays.",
|
||
"tags": [
|
||
"api"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"chemical-structure"
|
||
],
|
||
"organism": [],
|
||
"api": true,
|
||
"entities": [
|
||
"molecule",
|
||
"protein"
|
||
],
|
||
"documentation": "https://www.ebi.ac.uk/chembl/api/data/docs",
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://www.ebi.ac.uk/chembl/api/data/docs"
|
||
]
|
||
},
|
||
{
|
||
"id": "clinicaltrials_gov_api",
|
||
"name": "ClinicalTrials.gov API",
|
||
"type": "api",
|
||
"url": "https://clinicaltrials.gov/api/gui",
|
||
"description": "API for querying clinical trial metadata and results.",
|
||
"tags": [
|
||
"api"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"clinical"
|
||
],
|
||
"organism": [],
|
||
"api": true,
|
||
"entities": [
|
||
"disease",
|
||
"drug"
|
||
],
|
||
"documentation": "https://clinicaltrials.gov/data-api/api",
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://clinicaltrials.gov/data-api/api"
|
||
]
|
||
},
|
||
{
|
||
"id": "ensembl_rest_api",
|
||
"name": "Ensembl REST API",
|
||
"type": "api",
|
||
"url": "https://rest.ensembl.org/",
|
||
"description": "API for genomic annotations, variants, genes, and comparative genomics.",
|
||
"tags": [
|
||
"api"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"genomics"
|
||
],
|
||
"organism": [],
|
||
"api": true,
|
||
"entities": [
|
||
"gene",
|
||
"genome",
|
||
"transcript",
|
||
"variant"
|
||
],
|
||
"documentation": "https://rest.ensembl.org/",
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://rest.ensembl.org/"
|
||
]
|
||
},
|
||
{
|
||
"id": "kegg_rest_api",
|
||
"name": "KEGG REST API",
|
||
"type": "api",
|
||
"url": "https://www.kegg.jp/kegg/rest/keggapi.html",
|
||
"description": "API for accessing KEGG pathways, compounds, genes, and reactions.",
|
||
"tags": [
|
||
"api"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": true,
|
||
"entities": [
|
||
"compound",
|
||
"gene",
|
||
"pathway"
|
||
],
|
||
"documentation": "https://www.kegg.jp/kegg/rest/keggapi.html",
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://www.kegg.jp/kegg/rest/keggapi.html"
|
||
]
|
||
},
|
||
{
|
||
"id": "ncbi_e_utilities",
|
||
"name": "NCBI E-utilities",
|
||
"type": "api",
|
||
"url": "https://www.ncbi.nlm.nih.gov/books/NBK25501/",
|
||
"description": "Unified APIs for accessing NCBI databases (Gene, GEO, SRA, PubChem, etc).",
|
||
"tags": [
|
||
"api"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"genomics",
|
||
"transcriptomics"
|
||
],
|
||
"organism": [],
|
||
"api": true,
|
||
"entities": [
|
||
"gene",
|
||
"genome",
|
||
"protein",
|
||
"transcript",
|
||
"variant"
|
||
],
|
||
"documentation": "https://www.ncbi.nlm.nih.gov/books/NBK25501/",
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://www.ncbi.nlm.nih.gov/books/NBK25501/"
|
||
]
|
||
},
|
||
{
|
||
"id": "open_targets_platform_api",
|
||
"name": "Open Targets Platform API",
|
||
"type": "api",
|
||
"url": "https://platform.opentargets.org/api",
|
||
"description": "API for targetโdisease associations integrating genetics, genomics, and drug data.",
|
||
"tags": [
|
||
"api"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"genomics",
|
||
"knowledge-graph"
|
||
],
|
||
"organism": [],
|
||
"api": true,
|
||
"entities": [
|
||
"disease",
|
||
"drug",
|
||
"gene",
|
||
"variant"
|
||
],
|
||
"documentation": "https://platform.opentargets.org/api",
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://platform.opentargets.org/api"
|
||
]
|
||
},
|
||
{
|
||
"id": "pubmed_e_utilities_esearch_efetch",
|
||
"name": "PubMed E-utilities (esearch/efetch)",
|
||
"type": "api",
|
||
"url": "https://www.nlm.nih.gov/dataguide/edirect/esearch.html",
|
||
"description": "APIs for searching and retrieving biomedical literature from PubMed.",
|
||
"tags": [
|
||
"api"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": true,
|
||
"documentation": "https://www.ncbi.nlm.nih.gov/books/NBK25501/",
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://www.ncbi.nlm.nih.gov/books/NBK25501/"
|
||
]
|
||
},
|
||
{
|
||
"id": "uniprot_rest_api",
|
||
"name": "UniProt REST API",
|
||
"type": "api",
|
||
"url": "https://www.uniprot.org/help/api",
|
||
"description": "Programmatic access to protein sequence and functional annotation data.",
|
||
"tags": [
|
||
"api"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"protein-sequence",
|
||
"proteomics"
|
||
],
|
||
"organism": [],
|
||
"api": true,
|
||
"entities": [
|
||
"protein"
|
||
],
|
||
"documentation": "https://www.uniprot.org/help/api",
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://www.uniprot.org/help/api"
|
||
]
|
||
},
|
||
{
|
||
"id": "1000_genomes_project",
|
||
"name": "1000 Genomes Project",
|
||
"type": "benchmark",
|
||
"url": "https://www.internationalgenome.org/",
|
||
"description": "Reference panel of human genetic variation from 2,504 individuals across 26 populations.",
|
||
"tags": [
|
||
"benchmarks-and-datasets"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "bace",
|
||
"name": "BACE",
|
||
"type": "benchmark",
|
||
"url": "https://www.kaggle.com/datasets/gokturkkoch/bace",
|
||
"description": "Binary classification and regression dataset for ฮฒ-secretase 1 (BACE-1) inhibitor binding affinity.",
|
||
"tags": [
|
||
"benchmarks-and-datasets"
|
||
],
|
||
"tasks": [
|
||
"classification",
|
||
"regression"
|
||
],
|
||
"modalities": [
|
||
"chemical-structure"
|
||
],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"molecule",
|
||
"protein"
|
||
],
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://www.kaggle.com/datasets/gokturkkoch/bace"
|
||
]
|
||
},
|
||
{
|
||
"id": "beat_aml",
|
||
"name": "BEAT AML",
|
||
"type": "benchmark",
|
||
"url": "https://biodev.github.io/BeatAML2/",
|
||
"description": "Functional ex vivo drug sensitivity measurements paired with genomics for acute myeloid leukemia.",
|
||
"tags": [
|
||
"benchmarks-and-datasets"
|
||
],
|
||
"tasks": [
|
||
"drug-response-prediction"
|
||
],
|
||
"modalities": [
|
||
"genomics"
|
||
],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"cell",
|
||
"disease",
|
||
"drug",
|
||
"gene"
|
||
],
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://biodev.github.io/BeatAML2/"
|
||
]
|
||
},
|
||
{
|
||
"id": "bento",
|
||
"name": "Bento",
|
||
"type": "benchmark",
|
||
"url": "https://github.com/LigandPro/Bento",
|
||
"description": "Protein-ligand docking benchmark covering rigid, flexible, de novo, blind, induced-fit, and covalent docking tasks.",
|
||
"tags": [
|
||
"benchmarks-and-datasets"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "bindingdb_curated_sets",
|
||
"name": "BindingDB Curated Sets",
|
||
"type": "benchmark",
|
||
"url": "https://www.bindingdb.org/rwd/bind/chemsearch/marvin/SDFdownload.jsp?all_download=yes",
|
||
"description": "Curated binding affinity datasets for proteinโligand interaction benchmarking.",
|
||
"tags": [
|
||
"benchmarks-and-datasets"
|
||
],
|
||
"tasks": [
|
||
"drug-target-interaction"
|
||
],
|
||
"modalities": [
|
||
"chemical-structure"
|
||
],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"molecule",
|
||
"protein"
|
||
],
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://www.bindingdb.org/"
|
||
]
|
||
},
|
||
{
|
||
"id": "cancer_therapeutics_response_portal_ctrp",
|
||
"name": "Cancer Therapeutics Response Portal (CTRP)",
|
||
"type": "benchmark",
|
||
"url": "https://portals.broadinstitute.org/ctrp/",
|
||
"description": "Drug sensitivity profiles across ~900 cancer cell lines for >400 compounds.",
|
||
"tags": [
|
||
"benchmarks-and-datasets"
|
||
],
|
||
"tasks": [
|
||
"drug-response-prediction"
|
||
],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"cell",
|
||
"drug"
|
||
],
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://portals.broadinstitute.org/ctrp/"
|
||
]
|
||
},
|
||
{
|
||
"id": "clintox",
|
||
"name": "ClinTox",
|
||
"type": "benchmark",
|
||
"url": "https://tdcommons.ai/single_pred_tasks/tox/#clintox",
|
||
"description": "Clinical toxicity dataset contrasting FDA-approved drugs with those that failed clinical trials due to toxicity.",
|
||
"tags": [
|
||
"benchmarks-and-datasets"
|
||
],
|
||
"tasks": [
|
||
"classification"
|
||
],
|
||
"modalities": [
|
||
"clinical"
|
||
],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"drug"
|
||
],
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://tdcommons.ai/single_pred_tasks/tox/#clintox"
|
||
]
|
||
},
|
||
{
|
||
"id": "cptac_clinical_proteomic_tumor_analysis_consortium",
|
||
"name": "CPTAC (Clinical Proteomic Tumor Analysis Consortium)",
|
||
"type": "benchmark",
|
||
"url": "https://proteomics.cancer.gov/programs/cptac",
|
||
"description": "Multi-omic proteogenomic datasets for multiple cancer types linking proteomics with genomics.",
|
||
"tags": [
|
||
"benchmarks-and-datasets"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "crossdocked2020",
|
||
"name": "CrossDocked2020",
|
||
"type": "benchmark",
|
||
"url": "https://arxiv.org/abs/2001.01037",
|
||
"description": "Large-scale dataset for structure-based virtual screening.",
|
||
"tags": [
|
||
"benchmarks-and-datasets"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "dud_e_directory_of_useful_decoys_enhanced",
|
||
"name": "DUD-E (Directory of Useful Decoys, Enhanced)",
|
||
"type": "benchmark",
|
||
"url": "http://dude.docking.org/",
|
||
"description": "Structure-based virtual screening benchmark with active ligands and challenging decoy sets across diverse protein targets.",
|
||
"tags": [
|
||
"benchmarks-and-datasets"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "flip_fitness_landscape_inference_for_proteins",
|
||
"name": "FLIP (Fitness Landscape Inference for Proteins)",
|
||
"type": "benchmark",
|
||
"url": "https://github.com/J-SNACKKB/FLIP",
|
||
"description": "Benchmark collection of protein fitness landscape datasets for evaluating protein ML models.",
|
||
"tags": [
|
||
"benchmarks-and-datasets"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "guacamol",
|
||
"name": "GuacaMol",
|
||
"type": "benchmark",
|
||
"url": "https://github.com/BenevolentAI/guacamol",
|
||
"description": "Benchmark suite for generative molecular design models.",
|
||
"tags": [
|
||
"benchmarks-and-datasets"
|
||
],
|
||
"tasks": [
|
||
"molecular-generation"
|
||
],
|
||
"modalities": [
|
||
"chemical-structure"
|
||
],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"molecule"
|
||
],
|
||
"github": "https://github.com/BenevolentAI/guacamol",
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://github.com/BenevolentAI/guacamol"
|
||
]
|
||
},
|
||
{
|
||
"id": "hest_xenium_virtual_spatial_transcriptomics",
|
||
"name": "HEST Xenium virtual spatial transcriptomics",
|
||
"type": "benchmark",
|
||
"url": "https://huggingface.co/datasets/ratschlab/HEST_Xenium_virtual_spatial_transcriptomics",
|
||
"description": "DeepSpot-M predicted transcriptome-wide ST for 59 HEST-1k 10x Xenium samples (~13.3M cells) (gated). Paper: [DeepSpot-M](https://www.medrxiv.org/content/10.64898/2026.06.19.26356060v1).",
|
||
"tags": [
|
||
"benchmarks-and-datasets"
|
||
],
|
||
"tasks": [
|
||
"regression"
|
||
],
|
||
"modalities": [
|
||
"histopathology",
|
||
"spatial-transcriptomics",
|
||
"transcriptomics"
|
||
],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"cell",
|
||
"gene",
|
||
"tissue"
|
||
],
|
||
"documentation": "https://huggingface.co/datasets/ratschlab/HEST_Xenium_virtual_spatial_transcriptomics",
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://huggingface.co/datasets/ratschlab/HEST_Xenium_virtual_spatial_transcriptomics"
|
||
]
|
||
},
|
||
{
|
||
"id": "jump_cell_painting_datasets",
|
||
"name": "JUMP Cell Painting Datasets",
|
||
"type": "benchmark",
|
||
"url": "https://github.com/jump-cellpainting/datasets",
|
||
"description": "Consortium-scale cell imaging perturbation datasets (chemical and genetic) for phenotypic profiling and drug discovery research.",
|
||
"tags": [
|
||
"benchmarks-and-datasets"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "lincs_l1000",
|
||
"name": "LINCS L1000",
|
||
"type": "benchmark",
|
||
"url": "https://lincsproject.org/LINCS/tools/workflows/find-the-best-place-to-obtain-the-lincs-l1000-data",
|
||
"description": "Gene expression profiles (978 landmark genes) for >20,000 chemical and genetic perturbations across cell lines.",
|
||
"tags": [
|
||
"benchmarks-and-datasets"
|
||
],
|
||
"tasks": [
|
||
"perturbation-prediction"
|
||
],
|
||
"modalities": [
|
||
"transcriptomics"
|
||
],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"cell",
|
||
"compound",
|
||
"gene"
|
||
],
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://lincsproject.org/LINCS/tools/workflows/find-the-best-place-to-obtain-the-lincs-l1000-data"
|
||
]
|
||
},
|
||
{
|
||
"id": "moleculenet",
|
||
"name": "MoleculeNet",
|
||
"type": "benchmark",
|
||
"url": "http://moleculenet.ai/",
|
||
"description": "Benchmark datasets for molecular machine learning.",
|
||
"tags": [
|
||
"benchmarks-and-datasets"
|
||
],
|
||
"tasks": [
|
||
"classification",
|
||
"regression"
|
||
],
|
||
"modalities": [
|
||
"chemical-structure"
|
||
],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"molecule"
|
||
],
|
||
"github": "https://github.com/deepchem/moleculenet",
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://github.com/deepchem/moleculenet"
|
||
]
|
||
},
|
||
{
|
||
"id": "moses",
|
||
"name": "MOSES",
|
||
"type": "benchmark",
|
||
"url": "https://github.com/molecularsets/moses",
|
||
"description": "Benchmarking platform for molecular generation models.",
|
||
"tags": [
|
||
"benchmarks-and-datasets"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "ogb_open_graph_benchmark",
|
||
"name": "OGB (Open Graph Benchmark)",
|
||
"type": "benchmark",
|
||
"url": "https://ogb.stanford.edu/",
|
||
"description": "Large-scale graph ML benchmark suite including biological datasets such as ogbl-ppa (protein-protein associations) and ogbg-molhiv.",
|
||
"tags": [
|
||
"benchmarks-and-datasets"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "openbiolink",
|
||
"name": "OpenBioLink",
|
||
"type": "benchmark",
|
||
"url": "https://github.com/OpenBioLink/OpenBioLink",
|
||
"description": "Benchmark datasets for biological knowledge graph completion.",
|
||
"tags": [
|
||
"benchmarks-and-datasets"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "pharmgkb",
|
||
"name": "PharmGKB",
|
||
"type": "benchmark",
|
||
"url": "https://www.pharmgkb.org/",
|
||
"description": "Curated pharmacogenomics dataset linking genetic variants to drug response phenotypes across thousands of drugs.",
|
||
"tags": [
|
||
"benchmarks-and-datasets"
|
||
],
|
||
"tasks": [
|
||
"drug-response-prediction"
|
||
],
|
||
"modalities": [
|
||
"clinical",
|
||
"genomics"
|
||
],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"drug",
|
||
"gene",
|
||
"phenotype",
|
||
"variant"
|
||
],
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://www.pharmgkb.org/"
|
||
]
|
||
},
|
||
{
|
||
"id": "pk_db",
|
||
"name": "PK-DB",
|
||
"type": "benchmark",
|
||
"url": "https://pk-db.com/",
|
||
"description": "Open database of experimental pharmacokinetics (PK) and ADME data from clinical and preclinical studies.",
|
||
"tags": [
|
||
"benchmarks-and-datasets"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"clinical"
|
||
],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"drug"
|
||
],
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://pk-db.com/"
|
||
]
|
||
},
|
||
{
|
||
"id": "prism",
|
||
"name": "PRISM",
|
||
"type": "benchmark",
|
||
"url": "https://depmap.org/portal/prism/",
|
||
"description": "Cancer drug sensitivity profiling of >4,500 drugs across >900 cancer cell lines using pooled-cell-line barcoding.",
|
||
"tags": [
|
||
"benchmarks-and-datasets"
|
||
],
|
||
"tasks": [
|
||
"drug-response-prediction"
|
||
],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"cell",
|
||
"drug"
|
||
],
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://depmap.org/portal/prism/"
|
||
]
|
||
},
|
||
{
|
||
"id": "proteingym",
|
||
"name": "ProteinGym",
|
||
"type": "benchmark",
|
||
"url": "https://github.com/OATML-Markslab/ProteinGym",
|
||
"description": "Large-scale benchmark of deep mutational scanning assays for evaluating protein fitness landscape models.",
|
||
"tags": [
|
||
"benchmarks-and-datasets"
|
||
],
|
||
"tasks": [
|
||
"regression"
|
||
],
|
||
"modalities": [
|
||
"protein-sequence"
|
||
],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"protein"
|
||
],
|
||
"github": "https://github.com/OATML-Markslab/ProteinGym",
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://github.com/OATML-Markslab/ProteinGym"
|
||
]
|
||
},
|
||
{
|
||
"id": "qm9",
|
||
"name": "QM9",
|
||
"type": "benchmark",
|
||
"url": "https://figshare.com/collections/Quantum_chemistry_structures_and_properties_of_134_kilo_molecules/978904",
|
||
"description": "Quantum chemistry properties for 134K stable small organic molecules computed at DFT level.",
|
||
"tags": [
|
||
"benchmarks-and-datasets"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "scib_single_cell_integration_benchmarks",
|
||
"name": "scIB (Single-cell Integration Benchmarks)",
|
||
"type": "benchmark",
|
||
"url": "https://github.com/theislab/scib",
|
||
"description": "Comprehensive benchmarking framework for single-cell data integration methods.",
|
||
"tags": [
|
||
"benchmarks-and-datasets"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "scperturb",
|
||
"name": "scPerturb",
|
||
"type": "benchmark",
|
||
"url": "https://github.com/sanderlab/scPerturb",
|
||
"description": "Curated and continuously updated single-cell perturbation data resource spanning CRISPR and drug perturbation studies.",
|
||
"tags": [
|
||
"benchmarks-and-datasets"
|
||
],
|
||
"tasks": [
|
||
"perturbation-prediction"
|
||
],
|
||
"modalities": [
|
||
"single-cell-rna-seq"
|
||
],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"cell",
|
||
"drug",
|
||
"gene"
|
||
],
|
||
"github": "https://github.com/sanderlab/scPerturb",
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://github.com/sanderlab/scPerturb"
|
||
]
|
||
},
|
||
{
|
||
"id": "sider_side_effect_resource",
|
||
"name": "SIDER (Side Effect Resource)",
|
||
"type": "benchmark",
|
||
"url": "http://sideeffects.embl.de/",
|
||
"description": "Database of 1,430 approved drugs with their recorded adverse drug reactions across 27 system-organ classes.",
|
||
"tags": [
|
||
"benchmarks-and-datasets"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"clinical"
|
||
],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"drug",
|
||
"phenotype"
|
||
],
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"http://sideeffects.embl.de/"
|
||
]
|
||
},
|
||
{
|
||
"id": "tabula_muris",
|
||
"name": "Tabula Muris",
|
||
"type": "benchmark",
|
||
"url": "https://tabula-muris.ds.czbiohub.org/",
|
||
"description": "Comprehensive single-cell atlas of 20 mouse organs and tissues, enabling cross-tissue and cross-species comparisons.",
|
||
"tags": [
|
||
"benchmarks-and-datasets"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "tabula_sapiens",
|
||
"name": "Tabula Sapiens",
|
||
"type": "benchmark",
|
||
"url": "https://tabula-sapiens-portal.ds.czbiohub.org/",
|
||
"description": "Comprehensive human single-cell atlas of ~500K cells from 24 organs and tissues across multiple donors.",
|
||
"tags": [
|
||
"benchmarks-and-datasets"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "tape_tasks_assessing_protein_embeddings",
|
||
"name": "TAPE (Tasks Assessing Protein Embeddings)",
|
||
"type": "benchmark",
|
||
"url": "https://github.com/songlab-cal/tape",
|
||
"description": "Benchmark suite of five biologically meaningful semi-supervised learning tasks for evaluating protein representations.",
|
||
"tags": [
|
||
"benchmarks-and-datasets"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "tcga_virtual_spatial_transcriptomics_atlas",
|
||
"name": "TCGA virtual spatial transcriptomics atlas",
|
||
"type": "benchmark",
|
||
"url": "https://huggingface.co/datasets/ratschlab/TCGA_virtual_spatial_transcriptomics_atlas",
|
||
"description": "DeepSpot-M predicted transcriptome-wide ST for TCGA H&E (FF + FFPE; 28,664 slides / 32 cancer types; gated). Paper: [DeepSpot-M](https://www.medrxiv.org/content/10.64898/2026.06.19.26356060v1).",
|
||
"tags": [
|
||
"benchmarks-and-datasets"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "the_cancer_genome_atlas_tcga",
|
||
"name": "The Cancer Genome Atlas (TCGA)",
|
||
"type": "benchmark",
|
||
"url": "https://www.cancer.gov/about-nci/organization/ccg/research/structural-genomics/tcga",
|
||
"description": "Comprehensive multi-omics (genomics, transcriptomics, proteomics, methylation) dataset for 33 cancer types across ~11,000 patients.",
|
||
"tags": [
|
||
"benchmarks-and-datasets"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "therapeutics_data_commons_tdc",
|
||
"name": "Therapeutics Data Commons (TDC)",
|
||
"type": "benchmark",
|
||
"url": "https://tdcommons.ai/",
|
||
"description": "Unified benchmark suite covering ADMET, drug-target interaction, drug response, and more.",
|
||
"tags": [
|
||
"benchmarks-and-datasets"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "tox21",
|
||
"name": "Tox21",
|
||
"type": "benchmark",
|
||
"url": "https://tripod.nih.gov/tox21/challenge/",
|
||
"description": "12,707 compounds tested in 12 nuclear receptor and stress-response pathway biochemical assays for toxicity prediction.",
|
||
"tags": [
|
||
"benchmarks-and-datasets"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "uk_biobank",
|
||
"name": "UK Biobank",
|
||
"type": "benchmark",
|
||
"url": "https://www.ukbiobank.ac.uk/",
|
||
"description": "Large-scale biomedical database of ~500K participants with genetic, imaging, and health data for population genetics and disease studies.",
|
||
"tags": [
|
||
"benchmarks-and-datasets"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "10x_genomics_dataset",
|
||
"name": "10x Genomics Dataset",
|
||
"type": "database",
|
||
"url": "https://www.10xgenomics.com/resources/datasets",
|
||
"description": "Collection of single-cell datasets.",
|
||
"tags": [
|
||
"genome"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Genomics"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "alphafold_protein_structure_database",
|
||
"name": "AlphaFold Protein Structure Database",
|
||
"type": "database",
|
||
"url": "https://alphafold.ebi.ac.uk/api-docs",
|
||
"description": "3D protein structure predictions.",
|
||
"tags": [
|
||
"protein"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Protein"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "bindingdb",
|
||
"name": "BindingDB",
|
||
"type": "database",
|
||
"url": "https://www.bindingdb.org/rwd/bind/index.jsp",
|
||
"description": "Compounds and target database.",
|
||
"tags": [
|
||
"chemical-protein-interaction",
|
||
"interaction"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Protein",
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "biocyc",
|
||
"name": "BioCyc",
|
||
"type": "database",
|
||
"url": "https://biocyc.org/",
|
||
"description": "Collection of pathway/genome databases across thousands of organisms.",
|
||
"tags": [
|
||
"pathway"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Pathway"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "biogrid",
|
||
"name": "BioGRID",
|
||
"type": "database",
|
||
"url": "https://thebiogrid.org/",
|
||
"description": "Protein, genetic, and chemical interactions.",
|
||
"tags": [
|
||
"interaction",
|
||
"protein-protein-interaction"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Protein"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "cancer_cell_line_encyclopedia",
|
||
"name": "Cancer Cell Line Encyclopedia",
|
||
"type": "database",
|
||
"url": "https://sites.broadinstitute.org/ccle/",
|
||
"description": "Database of ~1000 cancer cell lines.",
|
||
"tags": [
|
||
"drug-cell-line-response",
|
||
"interaction"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Gene Expression",
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "catalogue_of_somatic_mutations_in_cancer_cosmic",
|
||
"name": "Catalogue Of Somatic Mutations In Cancer (COSMIC)",
|
||
"type": "database",
|
||
"url": "https://cancer.sanger.ac.uk/cosmic",
|
||
"description": "Resource on somatic mutations in cancers.",
|
||
"tags": [
|
||
"genome"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Genomics"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "cath_database",
|
||
"name": "CATH database",
|
||
"type": "database",
|
||
"url": "https://www.cathdb.info/",
|
||
"description": "Hierarchical classification of protein domain structures.",
|
||
"tags": [
|
||
"protein"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Protein"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "cbioportal",
|
||
"name": "cBioPortal",
|
||
"type": "database",
|
||
"url": "https://www.cbioportal.org/",
|
||
"description": "Cancer genomics database; aggregating many patient datasets.",
|
||
"tags": [
|
||
"genome"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Genomics"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "cellminer_cross_database_cellminercdb",
|
||
"name": "CellMiner Cross Database (CellMinerCDB)",
|
||
"type": "database",
|
||
"url": "https://discover.nci.nih.gov/cellminercdb/",
|
||
"description": "Integrates multiple cancer cell line databases.",
|
||
"tags": [
|
||
"drug-cell-line-response",
|
||
"interaction"
|
||
],
|
||
"tasks": [
|
||
"drug-response-prediction"
|
||
],
|
||
"modalities": [
|
||
"genomics"
|
||
],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"cell",
|
||
"drug",
|
||
"gene"
|
||
],
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://discover.nci.nih.gov/cellminercdb/"
|
||
]
|
||
},
|
||
{
|
||
"id": "chebi",
|
||
"name": "ChEBI",
|
||
"type": "database",
|
||
"url": "https://www.ebi.ac.uk/chebi/",
|
||
"description": "Database focused on small chemical compounds.",
|
||
"tags": [
|
||
"compound"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "chembl",
|
||
"name": "ChEMBL",
|
||
"type": "database",
|
||
"url": "https://www.ebi.ac.uk/chembl/",
|
||
"description": "Bioactive molecules with drug-like properties.",
|
||
"tags": [
|
||
"compound"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "chemspider",
|
||
"name": "ChemSpider",
|
||
"type": "database",
|
||
"url": "http://www.chemspider.com/",
|
||
"description": "Chemical structure database.",
|
||
"tags": [
|
||
"compound"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "clinicaltrials_gov",
|
||
"name": "ClinicalTrials.gov",
|
||
"type": "database",
|
||
"url": "https://clinicaltrials.gov/",
|
||
"description": "Privately and publicly funded clinical studies.",
|
||
"tags": [
|
||
"clinical-trial"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Clinical"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "comparative_toxicogenomics_database",
|
||
"name": "Comparative Toxicogenomics Database",
|
||
"type": "database",
|
||
"url": "http://ctdbase.org/",
|
||
"description": "Chemical-gene interactions, chemical-disease and gene-disease associations, chemical-phenotype associations.",
|
||
"tags": [
|
||
"drug-gene-interaction",
|
||
"interaction"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Gene",
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "critical_assessment_of_structure_prediction_casp",
|
||
"name": "Critical Assessment of Structure Prediction (CASP)",
|
||
"type": "database",
|
||
"url": "https://predictioncenter.org/",
|
||
"description": "Assessing methods for protein structure prediction.",
|
||
"tags": [
|
||
"protein"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Protein"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "cz_cellxgene",
|
||
"name": "CZ CELLxGENE",
|
||
"type": "database",
|
||
"url": "https://cellxgene.cziscience.com/",
|
||
"description": "Single-cell dataset repository and interactive explorer from the Chan Zuckerberg Initiative.",
|
||
"tags": [
|
||
"scrna"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Single Cell"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "davis_kinase_inhibitors_db",
|
||
"name": "Davis kinase inhibitors DB",
|
||
"type": "database",
|
||
"url": "http://staff.cs.utu.fi/~aijrinas/dti/",
|
||
"description": "Experimental kinase inhibitor binding affinity dataset for proteinโligand interaction research.",
|
||
"tags": [
|
||
"chemical-protein-interaction",
|
||
"interaction"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Protein",
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "dependency_map_depmap",
|
||
"name": "Dependency Map (DepMap)",
|
||
"type": "database",
|
||
"url": "https://depmap.org/portal/",
|
||
"description": "CRISPR-Cas9 screens in cancer cell lines.",
|
||
"tags": [
|
||
"genome"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Genomics"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "dgidb",
|
||
"name": "DGIdb",
|
||
"type": "database",
|
||
"url": "https://www.dgidb.org/",
|
||
"description": "Drug-gene interactions and the druggable genome.",
|
||
"tags": [
|
||
"drug-gene-interaction",
|
||
"interaction"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Gene",
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "diseases",
|
||
"name": "DISEASES",
|
||
"type": "database",
|
||
"url": "https://diseases.jensenlab.org/",
|
||
"description": "Geneโdisease association database integrating evidence from text mining, curated databases, and experimental data.",
|
||
"tags": [
|
||
"disease"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Disease"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "disgenet",
|
||
"name": "DisGeNET",
|
||
"type": "database",
|
||
"url": "https://www.disgenet.org/",
|
||
"description": "Database of gene-disease associations integrating expert-curated and GWAS data.",
|
||
"tags": [
|
||
"disease"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Disease"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "drkg",
|
||
"name": "DRKG",
|
||
"type": "database",
|
||
"url": "https://github.com/gnn4dr/DRKG",
|
||
"description": "Large-scale biological knowledge graph for drug discovery.",
|
||
"tags": [
|
||
"interaction",
|
||
"knowledge-graph"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Knowledge Graph"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "drug_mechanism_database_drugmechdb",
|
||
"name": "Drug Mechanism Database (DrugMechDB)",
|
||
"type": "database",
|
||
"url": "https://github.com/SuLab/DrugMechDB/tree/2.0.1",
|
||
"description": "Mechanisms of action from drug to disease.",
|
||
"tags": [
|
||
"interaction",
|
||
"knowledge-graph"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Knowledge Graph"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "drug_repurposing_hub",
|
||
"name": "Drug Repurposing Hub",
|
||
"type": "database",
|
||
"url": "https://repo-hub.broadinstitute.org/repurposing#download-data",
|
||
"description": "Collections of drug repurposing data (drug, MoA, target, etc).",
|
||
"tags": [
|
||
"compound"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "drugbank",
|
||
"name": "DrugBank",
|
||
"type": "database",
|
||
"url": "https://go.drugbank.com/",
|
||
"description": "Database of drugs and targets (University of Alberta).",
|
||
"tags": [
|
||
"disease"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"chemical-structure"
|
||
],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"disease",
|
||
"drug",
|
||
"protein"
|
||
],
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://go.drugbank.com/"
|
||
]
|
||
},
|
||
{
|
||
"id": "drugcentral",
|
||
"name": "DrugCentral",
|
||
"type": "database",
|
||
"url": "http://drugcentral.org/",
|
||
"description": "Online drug compendium with drug mode of action and indication information.",
|
||
"tags": [
|
||
"compound"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "drugtargetcommons",
|
||
"name": "DrugTargetCommons",
|
||
"type": "database",
|
||
"url": "https://drugtargetcommons.fimm.fi/",
|
||
"description": "Community platform for curating and integrating experimental bioactivity data across drugs and targets.",
|
||
"tags": [
|
||
"compound"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "encode",
|
||
"name": "ENCODE",
|
||
"type": "database",
|
||
"url": "https://www.encodeproject.org/",
|
||
"description": "Encyclopedia of DNA Elements; regulatory and functional genomic elements across the genome.",
|
||
"tags": [
|
||
"genome"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Genomics"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "ensembl",
|
||
"name": "Ensembl",
|
||
"type": "database",
|
||
"url": "https://www.ensembl.org/",
|
||
"description": "Genome browser and annotation database for vertebrate and other eukaryotic genomes.",
|
||
"tags": [
|
||
"genome"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Genomics"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "eu_drug_regulating_authorities_clinical_trials_db_eudract",
|
||
"name": "EU Drug Regulating Authorities Clinical Trials DB (EudraCT)",
|
||
"type": "database",
|
||
"url": "https://eudract.ema.europa.eu/",
|
||
"description": "European clinical trial database.",
|
||
"tags": [
|
||
"clinical-trial"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Clinical"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "fantom5",
|
||
"name": "FANTOM5",
|
||
"type": "database",
|
||
"url": "https://fantom.gsc.riken.jp/5/",
|
||
"description": "Functional annotation of mammalian genome; comprehensive atlas of active enhancers, promoters, and transcription start sites across human and mouse cell types.",
|
||
"tags": [
|
||
"genome"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Genomics"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "genbank",
|
||
"name": "GenBank",
|
||
"type": "database",
|
||
"url": "https://www.ncbi.nlm.nih.gov/genbank/",
|
||
"description": "NCBI's database of genetic sequences.",
|
||
"tags": [
|
||
"genome"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Genomics"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "gene_expression_omnibus",
|
||
"name": "Gene Expression Omnibus",
|
||
"type": "database",
|
||
"url": "https://www.ncbi.nlm.nih.gov/geo/",
|
||
"description": "Public functional genomics database.",
|
||
"tags": [
|
||
"scrna"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Single Cell"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "genomics_of_drug_sensitivity_in_cancer_gdsc",
|
||
"name": "Genomics of Drug Sensitivity in Cancer (GDSC)",
|
||
"type": "database",
|
||
"url": "https://www.cancerrxgene.org/",
|
||
"description": "Drug sensitivity for ~1000 human cancer cell lines and hundreds of compounds.",
|
||
"tags": [
|
||
"benchmarks-and-datasets",
|
||
"drug-cell-line-response",
|
||
"interaction"
|
||
],
|
||
"tasks": [
|
||
"drug-response-prediction"
|
||
],
|
||
"modalities": [
|
||
"genomics"
|
||
],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"cell",
|
||
"drug",
|
||
"gene"
|
||
],
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://www.cancerrxgene.org/"
|
||
]
|
||
},
|
||
{
|
||
"id": "gnomad",
|
||
"name": "gnomAD",
|
||
"type": "database",
|
||
"url": "https://gnomad.broadinstitute.org/",
|
||
"description": "Genome Aggregation Database; genetic variation from large-scale sequencing projects.",
|
||
"tags": [
|
||
"genome"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Genomics"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "hetionet",
|
||
"name": "Hetionet",
|
||
"type": "database",
|
||
"url": "https://github.com/hetio/hetionet",
|
||
"description": "Heterogeneous network integrating genes, diseases, drugs, pathways, and more.",
|
||
"tags": [
|
||
"interaction",
|
||
"knowledge-graph"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Knowledge Graph"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "hippie",
|
||
"name": "HIPPIE",
|
||
"type": "database",
|
||
"url": "http://cbdm-01.zdv.uni-mainz.de/~mschaefer/hippie/",
|
||
"description": "Human protein-protein interaction database.",
|
||
"tags": [
|
||
"interaction",
|
||
"protein-protein-interaction"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Protein"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "hmdb_human_metabolome_database",
|
||
"name": "HMDB (Human Metabolome Database)",
|
||
"type": "database",
|
||
"url": "https://hmdb.ca/",
|
||
"description": "Comprehensive database of small molecule metabolites found in the human body.",
|
||
"tags": [
|
||
"compound"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "human_cell_atlas",
|
||
"name": "Human Cell Atlas",
|
||
"type": "database",
|
||
"url": "https://www.humancellatlas.org/",
|
||
"description": "Open global atlas of all cells in the human body.",
|
||
"tags": [
|
||
"scrna"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Single Cell"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "human_genome_resources_at_ncbi",
|
||
"name": "Human Genome Resources at NCBI",
|
||
"type": "database",
|
||
"url": "https://www.ncbi.nlm.nih.gov/projects/genome/guide/human/index.shtml",
|
||
"description": "Database for genomics, proteomics, transcriptomics, and systems biology.",
|
||
"tags": [
|
||
"genome"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Genomics"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "human_phenotype_ontology_hpo",
|
||
"name": "Human Phenotype Ontology (HPO)",
|
||
"type": "database",
|
||
"url": "https://hpo.jax.org/",
|
||
"description": "Standardized vocabulary of phenotypic abnormalities in human disease, linking genes, variants, and clinical features.",
|
||
"tags": [
|
||
"disease"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Disease"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "icd10",
|
||
"name": "ICD10",
|
||
"type": "database",
|
||
"url": "https://icd.who.int/browse10/2019/en",
|
||
"description": "International Classification of Diseases, 10th revision.",
|
||
"tags": [
|
||
"clinical-trial"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Clinical"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "intact",
|
||
"name": "IntAct",
|
||
"type": "database",
|
||
"url": "https://www.ebi.ac.uk/intact/home",
|
||
"description": "Open-source molecular interaction database and analysis system from EMBL-EBI.",
|
||
"tags": [
|
||
"interaction",
|
||
"protein-protein-interaction"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Protein"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "interpro",
|
||
"name": "InterPro",
|
||
"type": "database",
|
||
"url": "https://www.ebi.ac.uk/interpro/",
|
||
"description": "Protein families, domains, and functional sites database integrating 14 member databases including Pfam and PROSITE.",
|
||
"tags": [
|
||
"protein"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Protein"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "jaspar",
|
||
"name": "JASPAR",
|
||
"type": "database",
|
||
"url": "http://jaspar.genereg.net/",
|
||
"description": "Database of transcription factor binding profiles.",
|
||
"tags": [
|
||
"genome"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Genomics"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "kegg_compound",
|
||
"name": "KEGG COMPOUND",
|
||
"type": "database",
|
||
"url": "https://www.genome.jp/kegg/compound/",
|
||
"description": "Collection of small molecules and biopolymers.",
|
||
"tags": [
|
||
"compound"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "kegg_drug",
|
||
"name": "KEGG DRUG",
|
||
"type": "database",
|
||
"url": "https://www.genome.jp/kegg/drug/",
|
||
"description": "Comprehensive, approved drug information.",
|
||
"tags": [
|
||
"disease"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Disease"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "kegg_pathway",
|
||
"name": "KEGG PATHWAY",
|
||
"type": "database",
|
||
"url": "https://www.genome.jp/kegg/pathway.html",
|
||
"description": "Collection of pathway maps.",
|
||
"tags": [
|
||
"pathway"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Pathway"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "kinase_inhibitor_bioactivity_data_kiba",
|
||
"name": "Kinase Inhibitor Bioactivity Data (KIBA)",
|
||
"type": "database",
|
||
"url": "https://janeliascicomp.github.io/KIBA/",
|
||
"description": "Integrated bioactivity scores for kinase inhibitors combining Ki, Kd, and IC50 measurements.",
|
||
"tags": [
|
||
"chemical-protein-interaction",
|
||
"interaction"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Protein",
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "lipid_maps",
|
||
"name": "LIPID MAPS",
|
||
"type": "database",
|
||
"url": "https://www.lipidmaps.org/databases/lmsd/overview",
|
||
"description": "Database of lipids.",
|
||
"tags": [
|
||
"compound"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "massbank",
|
||
"name": "MassBank",
|
||
"type": "database",
|
||
"url": "http://www.massbank.jp/",
|
||
"description": "Open source databases and tools for mass spectrometry reference spectra.",
|
||
"tags": [
|
||
"mass-spectra"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Mass Spectra"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "mgnify",
|
||
"name": "MGnify",
|
||
"type": "database",
|
||
"url": "https://www.ebi.ac.uk/metagenomics/",
|
||
"description": "Resource for metagenomic and metatranscriptomic data.",
|
||
"tags": [
|
||
"genome"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Genomics"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "mimic_iv",
|
||
"name": "MIMIC-IV",
|
||
"type": "database",
|
||
"url": "https://mimic.mit.edu/",
|
||
"description": "Freely accessible critical care database.",
|
||
"tags": [
|
||
"clinical-trial"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Clinical"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "mirbase",
|
||
"name": "miRBase",
|
||
"type": "database",
|
||
"url": "https://www.mirbase.org/",
|
||
"description": "Reference repository for microRNA gene annotations, sequences, and experimentally validated targets.",
|
||
"tags": [
|
||
"gene-regulatory-network",
|
||
"interaction"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Gene Expression"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "mona_massbank_of_north_america",
|
||
"name": "MoNA MassBank of North America",
|
||
"type": "database",
|
||
"url": "https://mona.fiehnlab.ucdavis.edu/",
|
||
"description": "Meta-database of metabolite mass spectra, metadata, and associated compounds.",
|
||
"tags": [
|
||
"mass-spectra"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Mass Spectra"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "msigdb_molecular_signatures_database",
|
||
"name": "MSigDB (Molecular Signatures Database)",
|
||
"type": "database",
|
||
"url": "https://www.gsea-msigdb.org/gsea/msigdb",
|
||
"description": "Curated gene sets derived from pathways and biological processes.",
|
||
"tags": [
|
||
"pathway"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Pathway"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "nci60",
|
||
"name": "NCI60",
|
||
"type": "database",
|
||
"url": "https://dtp.cancer.gov/discovery_development/nci-60/",
|
||
"description": "Focuses on 60 cancer cell lines and many drugs.",
|
||
"tags": [
|
||
"benchmarks-and-datasets",
|
||
"drug-cell-line-response",
|
||
"interaction"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Gene Expression",
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "nextprot",
|
||
"name": "NeXtProt",
|
||
"type": "database",
|
||
"url": "https://www.nextprot.org/",
|
||
"description": "Expert knowledge base on human proteins with deep functional annotation, complementary to UniProt.",
|
||
"tags": [
|
||
"protein"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Protein"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "oadb_observed_antibody_space_database",
|
||
"name": "OADB (Observed Antibody Space Database)",
|
||
"type": "database",
|
||
"url": "http://opig.stats.ox.ac.uk/webapps/oas/",
|
||
"description": "Database of antibody sequences from immune repertoire sequencing.",
|
||
"tags": [
|
||
"protein"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Protein"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "omim_online_mendelian_inheritance_in_man",
|
||
"name": "OMIM (Online Mendelian Inheritance in Man)",
|
||
"type": "database",
|
||
"url": "https://www.omim.org/",
|
||
"description": "Comprehensive database of human genes and genetic disorders.",
|
||
"tags": [
|
||
"disease"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Disease"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "omnipath",
|
||
"name": "OmniPath",
|
||
"type": "database",
|
||
"url": "https://omnipathdb.org/",
|
||
"description": "Comprehensive resource integrating protein interactions, signaling pathways, gene regulatory networks, and miRNA targets from over 100 databases.",
|
||
"tags": [
|
||
"pathway"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Pathway"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "open_targets_platform",
|
||
"name": "Open Targets Platform",
|
||
"type": "database",
|
||
"url": "https://platform.opentargets.org/",
|
||
"description": "Systematic target identification and prioritization platform integrating genetics, genomics, and drug data for drug discovery.",
|
||
"tags": [
|
||
"disease"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Disease"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "pathwaycommons",
|
||
"name": "PathwayCommons",
|
||
"type": "database",
|
||
"url": "https://www.pathwaycommons.org/",
|
||
"description": "Database of pathways and interactions.",
|
||
"tags": [
|
||
"pathway"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Pathway"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "pdbbind",
|
||
"name": "PDBBind",
|
||
"type": "database",
|
||
"url": "https://www.pdbbind-plus.org.cn/",
|
||
"description": "Binding affinity data for biomolecular complexes.",
|
||
"tags": [
|
||
"chemical-protein-interaction",
|
||
"interaction"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Protein",
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "pfam",
|
||
"name": "Pfam",
|
||
"type": "database",
|
||
"url": "https://www.ebi.ac.uk/interpro/entry/pfam/",
|
||
"description": "Database of protein families described by multiple sequence alignments and hidden Markov models.",
|
||
"tags": [
|
||
"protein"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Protein"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "primekg",
|
||
"name": "PrimeKG",
|
||
"type": "database",
|
||
"url": "https://github.com/mims-harvard/PrimeKG",
|
||
"description": "Multi-modal precision medicine knowledge graph integrating clinical, genetic, and drug data.",
|
||
"tags": [
|
||
"interaction",
|
||
"knowledge-graph"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Knowledge Graph"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "protein_data_bank_pdb",
|
||
"name": "PROTEIN DATA BANK (PDB)",
|
||
"type": "database",
|
||
"url": "https://www.rcsb.org/",
|
||
"description": "3D structures of proteins, nucleic acids, complexes.",
|
||
"tags": [
|
||
"protein"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Protein"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "pubchem",
|
||
"name": "PubChem",
|
||
"type": "database",
|
||
"url": "https://pubchem.ncbi.nlm.nih.gov/",
|
||
"description": "One of the largest chemical databases (compounds, genes, and proteins).",
|
||
"tags": [
|
||
"compound"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "rcsb_protein_data_bank",
|
||
"name": "RCSB Protein Data Bank",
|
||
"type": "database",
|
||
"url": "https://www.rcsb.org/",
|
||
"description": "Repository for structural data of biological molecules.",
|
||
"tags": [
|
||
"protein"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Protein"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "reactome",
|
||
"name": "Reactome",
|
||
"type": "database",
|
||
"url": "https://reactome.org/",
|
||
"description": "Expert-curated, peer-reviewed pathway database with detailed reaction mechanisms.",
|
||
"tags": [
|
||
"pathway"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Pathway"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "regnetwork",
|
||
"name": "RegNetwork",
|
||
"type": "database",
|
||
"url": "http://www.regnetworkweb.org/",
|
||
"description": "Database of gene regulatory networks covering transcription factorโtarget gene and miRNAโgene interaction data across multiple species.",
|
||
"tags": [
|
||
"gene-regulatory-network",
|
||
"interaction"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Gene Expression"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "rfam",
|
||
"name": "Rfam",
|
||
"type": "database",
|
||
"url": "https://rfam.org/",
|
||
"description": "Database of RNA families with sequence alignments and consensus structures.",
|
||
"tags": [
|
||
"genome"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Genomics"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "rhea",
|
||
"name": "Rhea",
|
||
"type": "database",
|
||
"url": "https://www.rhea-db.org/",
|
||
"description": "Database of chemical reactions.",
|
||
"tags": [
|
||
"compound"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "roadmap_epigenomics",
|
||
"name": "ROADMAP Epigenomics",
|
||
"type": "database",
|
||
"url": "http://www.roadmapepigenomics.org/",
|
||
"description": "Reference epigenome maps for 111 primary human cell types and tissues, including histone modifications, chromatin accessibility, and DNA methylation.",
|
||
"tags": [
|
||
"genome"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Genomics"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "sabdab",
|
||
"name": "SAbDab",
|
||
"type": "database",
|
||
"url": "https://opig.stats.ox.ac.uk/webapps/sabdab-sabpred/sabdab",
|
||
"description": "Structural Antibody Database containing all antibody structures in the PDB.",
|
||
"tags": [
|
||
"protein"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Protein"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "signor_2_0",
|
||
"name": "SIGNOR 2.0",
|
||
"type": "database",
|
||
"url": "https://signor.uniroma2.it/",
|
||
"description": "Database of causal signaling interactions and pathways, with signed and directed relationships between proteins.",
|
||
"tags": [
|
||
"pathway"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Pathway"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "single_cell_expression_atlas",
|
||
"name": "Single Cell Expression Atlas",
|
||
"type": "database",
|
||
"url": "https://www.ebi.ac.uk/gxa/sc/home",
|
||
"description": "Public database for single-cell RNA.",
|
||
"tags": [
|
||
"scrna"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Single Cell"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "single_cell_portal",
|
||
"name": "Single Cell PORTAL",
|
||
"type": "database",
|
||
"url": "https://singlecell.broadinstitute.org/single_cell",
|
||
"description": "Public database for single-cell RNA.",
|
||
"tags": [
|
||
"scrna"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Single Cell"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "snap",
|
||
"name": "SNAP",
|
||
"type": "database",
|
||
"url": "https://snap.stanford.edu/biodata/datasets/10002/10002-ChG-Miner.html",
|
||
"description": "Dataset of drug-gene interactions.",
|
||
"tags": [
|
||
"drug-gene-interaction",
|
||
"interaction"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Gene",
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "stitch",
|
||
"name": "STITCH",
|
||
"type": "database",
|
||
"url": "http://stitch.embl.de/",
|
||
"description": "Chemical-protein interactions.",
|
||
"tags": [
|
||
"chemical-protein-interaction",
|
||
"interaction"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Protein",
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "string",
|
||
"name": "STRING",
|
||
"type": "database",
|
||
"url": "https://string-db.org/",
|
||
"description": "PPI networks for multiple organisms.",
|
||
"tags": [
|
||
"interaction",
|
||
"protein-protein-interaction"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"knowledge-graph",
|
||
"proteomics"
|
||
],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"protein"
|
||
],
|
||
"documentation": "https://string-db.org/help/api/",
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://string-db.org/",
|
||
"https://string-db.org/help/api/"
|
||
]
|
||
},
|
||
{
|
||
"id": "the_genotype_tissue_expression_gtex",
|
||
"name": "The Genotype-Tissue Expression (GTEx)",
|
||
"type": "database",
|
||
"url": "https://gtexportal.org/home/",
|
||
"description": "Human gene expression and regulation resource.",
|
||
"tags": [
|
||
"genome"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Genomics"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "the_human_protein_atlas",
|
||
"name": "THE HUMAN PROTEIN ATLAS",
|
||
"type": "database",
|
||
"url": "https://www.proteinatlas.org/",
|
||
"description": "Comprehensive human protein database (cells, tissues, organs).",
|
||
"tags": [
|
||
"protein"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Protein"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "therapeutic_target_database",
|
||
"name": "Therapeutic Target Database",
|
||
"type": "database",
|
||
"url": "https://idrblab.net/ttd/full-data-download",
|
||
"description": "Drug-target, target-disease, and drug-disease datasets.",
|
||
"tags": [
|
||
"compound"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "trrust_v2",
|
||
"name": "TRRUST v2",
|
||
"type": "database",
|
||
"url": "https://www.grnpedia.org/trrust/",
|
||
"description": "Manually curated database of human and mouse transcriptional regulatory interactions between transcription factors and their target genes, expanded with literature-derived evidence.",
|
||
"tags": [
|
||
"gene-regulatory-network",
|
||
"interaction"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Gene Expression"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "ucsc_genome_browser",
|
||
"name": "UCSC Genome Browser",
|
||
"type": "database",
|
||
"url": "https://genome.ucsc.edu/",
|
||
"description": "UCSC's genome browser.",
|
||
"tags": [
|
||
"genome"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Genomics"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "uniclust",
|
||
"name": "Uniclust",
|
||
"type": "database",
|
||
"url": "https://uniclust.mmseqs.com/",
|
||
"description": "Clustered protein sequence databases.",
|
||
"tags": [
|
||
"protein"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Protein"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "uniprot",
|
||
"name": "UniProt",
|
||
"type": "database",
|
||
"url": "https://www.uniprot.org/",
|
||
"description": "Functional information on proteins.",
|
||
"tags": [
|
||
"protein"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Protein"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "uniref",
|
||
"name": "UniRef",
|
||
"type": "database",
|
||
"url": "https://www.uniprot.org/uniref/",
|
||
"description": "Non-redundant sequence database clustering UniProtKB entries at multiple sequence identity thresholds.",
|
||
"tags": [
|
||
"protein"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Protein"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "wikipathways",
|
||
"name": "WikiPathways",
|
||
"type": "database",
|
||
"url": "https://wikipathways.org/",
|
||
"description": "Database of biological pathways.",
|
||
"tags": [
|
||
"pathway"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Pathway"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "zinc_ligand_discovery_database",
|
||
"name": "ZINC ligand discovery database",
|
||
"type": "database",
|
||
"url": "https://zinc.docking.org/",
|
||
"description": "Free database of commercially-available compounds for virtual screening.",
|
||
"tags": [
|
||
"compound"
|
||
],
|
||
"tasks": [],
|
||
"modalities": [
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "aestetik",
|
||
"name": "AESTETIK",
|
||
"type": "model",
|
||
"url": "https://github.com/ratschlab/aestetik",
|
||
"description": "Autoencoder for spatial transcriptomics representation learning using topology and histology image knowledge.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"single-cell-foundation-models",
|
||
"spatial-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"representation-learning"
|
||
],
|
||
"modalities": [
|
||
"histopathology",
|
||
"spatial-transcriptomics"
|
||
],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"cell",
|
||
"gene",
|
||
"tissue"
|
||
],
|
||
"methods": [
|
||
"autoencoder"
|
||
],
|
||
"github": "https://github.com/ratschlab/aestetik",
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://github.com/ratschlab/aestetik"
|
||
]
|
||
},
|
||
{
|
||
"id": "ai4chem_chemllm_7b_chat",
|
||
"name": "AI4Chem/ChemLLM-7B-Chat",
|
||
"type": "model",
|
||
"url": "https://huggingface.co/AI4Chem/ChemLLM-7B-Chat",
|
||
"description": "LLM for chemical & molecular science.",
|
||
"tags": [
|
||
"llm-for-biology"
|
||
],
|
||
"tasks": [
|
||
"Language Modeling"
|
||
],
|
||
"modalities": [
|
||
"Text"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "alphafold3",
|
||
"name": "AlphaFold3",
|
||
"type": "model",
|
||
"url": "https://github.com/google-deepmind/alphafold3",
|
||
"description": "Predicts structures of proteins, nucleic acids, small molecules, and their complexes.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"protein-foundation-models",
|
||
"protein-structure-prediction-and-design"
|
||
],
|
||
"tasks": [
|
||
"structure-prediction"
|
||
],
|
||
"modalities": [
|
||
"molecular-structure",
|
||
"protein-sequence"
|
||
],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"molecule",
|
||
"protein",
|
||
"protein-complex"
|
||
],
|
||
"methods": [
|
||
"diffusion"
|
||
],
|
||
"github": "https://github.com/google-deepmind/alphafold3",
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://github.com/google-deepmind/alphafold3"
|
||
]
|
||
},
|
||
{
|
||
"id": "ankh",
|
||
"name": "Ankh",
|
||
"type": "model",
|
||
"url": "https://github.com/agemagician/Ankh",
|
||
"description": "Efficient protein language model optimized for downstream prediction tasks including secondary structure, localization, and function annotation.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"pre-trained-embedding",
|
||
"protein-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"Foundation Model"
|
||
],
|
||
"modalities": [
|
||
"Protein"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "babel",
|
||
"name": "BABEL",
|
||
"type": "model",
|
||
"url": "https://github.com/wukevin/babel",
|
||
"description": "Cross-modality translation model enabling prediction between scRNA-seq and scATAC-seq profiles without requiring paired single-cell measurements.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"multi-omics-foundation-models",
|
||
"single-cell-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"Foundation Model"
|
||
],
|
||
"modalities": [
|
||
"Multi-Omics",
|
||
"Single Cell"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "basenji",
|
||
"name": "Basenji",
|
||
"type": "model",
|
||
"url": "https://github.com/calico/basenji",
|
||
"description": "Sequential regulatory activity prediction from DNA sequences.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"genomics-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"Foundation Model"
|
||
],
|
||
"modalities": [
|
||
"Genomics"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "biogpt",
|
||
"name": "BioGPT",
|
||
"type": "model",
|
||
"url": "https://github.com/microsoft/BioGPT",
|
||
"description": "LLM for biomedical text generation.",
|
||
"tags": [
|
||
"llm-for-biology"
|
||
],
|
||
"tasks": [
|
||
"Language Modeling"
|
||
],
|
||
"modalities": [
|
||
"Text"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "biomedclip",
|
||
"name": "BiomedCLIP",
|
||
"type": "model",
|
||
"url": "https://huggingface.co/microsoft/BiomedCLIP-PubMedBERT_256-vit_g_14",
|
||
"description": "CLIP-based vision-language foundation model for biomedical images and text trained on PubMed figureโcaption pairs.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"multi-modal-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"Foundation Model"
|
||
],
|
||
"modalities": [
|
||
"Multi-Modal"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "biomedlm",
|
||
"name": "BioMedLM",
|
||
"type": "model",
|
||
"url": "https://huggingface.co/stanford-crfm/BioMedLM",
|
||
"description": "2.7B parameter GPT-2-style language model trained exclusively on biomedical literature from PubMed for biomedical question answering and text generation.",
|
||
"tags": [
|
||
"llm-for-biology"
|
||
],
|
||
"tasks": [
|
||
"Language Modeling"
|
||
],
|
||
"modalities": [
|
||
"Text"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "boltz_1",
|
||
"name": "Boltz-1",
|
||
"type": "model",
|
||
"url": "https://github.com/jwohlwend/boltz",
|
||
"description": "Open-source all-atom biomolecular structure prediction model for proteins, nucleic acids, small molecules, and their complexes achieving AlphaFold3-level accuracy.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"protein-foundation-models",
|
||
"protein-structure-prediction-and-design"
|
||
],
|
||
"tasks": [
|
||
"Foundation Model",
|
||
"Protein Structure Prediction"
|
||
],
|
||
"modalities": [
|
||
"Protein"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "borzoi",
|
||
"name": "Borzoi",
|
||
"type": "model",
|
||
"url": "https://github.com/calico/borzoi",
|
||
"description": "Extended successor to Enformer for predicting RNA-seq coverage from long genomic sequence windows (524 kb) with improved resolution.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"genomics-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"Foundation Model"
|
||
],
|
||
"modalities": [
|
||
"Genomics"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "bulkformer",
|
||
"name": "BulkFormer",
|
||
"type": "model",
|
||
"url": "https://github.com/KangBoming/BulkFormer",
|
||
"description": "Foundation model for bulk RNA-seq data; learns general transcriptomic representations.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"single-cell-foundation-models",
|
||
"transcriptomics-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"Foundation Model"
|
||
],
|
||
"modalities": [
|
||
"Single Cell",
|
||
"Transcriptomics"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "caduceus",
|
||
"name": "Caduceus",
|
||
"type": "model",
|
||
"url": "https://github.com/kuleshov-group/caduceus",
|
||
"description": "Bidirectional equivariant long-range DNA sequence model based on Mamba.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"genomics-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"Foundation Model"
|
||
],
|
||
"modalities": [
|
||
"Genomics"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "cancerfoundation",
|
||
"name": "CancerFoundation",
|
||
"type": "model",
|
||
"url": "https://github.com/BoevaLab/CancerFoundation",
|
||
"description": "Single-cell RNA-seq foundation model trained exclusively on a curated dataset of malignant cells to learn cancer-specific embeddings.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"single-cell-foundation-models",
|
||
"transcriptomics-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"Foundation Model"
|
||
],
|
||
"modalities": [
|
||
"Single Cell",
|
||
"Transcriptomics"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "cassia",
|
||
"name": "CASSIA",
|
||
"type": "model",
|
||
"url": "https://github.com/ElliotXie/CASSIA",
|
||
"description": "Multi-agent LLM for reference-free, interpretable cell-type annotation of single-cell RNA-seq data, with dedicated annotation, validation, scoring, and reporting agents.",
|
||
"tags": [
|
||
"llm-for-biology"
|
||
],
|
||
"tasks": [
|
||
"Language Modeling"
|
||
],
|
||
"modalities": [
|
||
"Text"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "cellot",
|
||
"name": "CellOT",
|
||
"type": "model",
|
||
"url": "https://github.com/bunnech/cellot",
|
||
"description": "Neural optimal transport framework for predicting single-cell responses to drug and genetic perturbations.",
|
||
"tags": [
|
||
"drug-discovery",
|
||
"drug-perturbation"
|
||
],
|
||
"tasks": [
|
||
"Drug Discovery",
|
||
"Drug Perturbation"
|
||
],
|
||
"modalities": [
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "cellplm",
|
||
"name": "CellPLM",
|
||
"type": "model",
|
||
"url": "https://github.com/OmicsML/CellPLM",
|
||
"description": "Cell pre-trained language model with inter-cell transformer architecture for diverse single-cell analysis tasks.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"single-cell-foundation-models",
|
||
"transcriptomics-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"foundation-model-pretraining",
|
||
"representation-learning"
|
||
],
|
||
"modalities": [
|
||
"single-cell-rna-seq",
|
||
"transcriptomics"
|
||
],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"cell",
|
||
"gene"
|
||
],
|
||
"methods": [
|
||
"self-supervised-learning",
|
||
"transformer"
|
||
],
|
||
"year": 2023,
|
||
"github": "https://github.com/OmicsML/CellPLM",
|
||
"paper": "https://www.biorxiv.org/content/10.1101/2023.10.03.560734v1",
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://github.com/OmicsML/CellPLM",
|
||
"https://www.biorxiv.org/content/10.1101/2023.10.03.560734v1"
|
||
]
|
||
},
|
||
{
|
||
"id": "chai_1",
|
||
"name": "Chai-1",
|
||
"type": "model",
|
||
"url": "https://github.com/chaidiscovery/chai-lab",
|
||
"description": "Unified molecular structure prediction model covering proteins, nucleic acids, small molecules, and complexes.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"protein-foundation-models",
|
||
"protein-structure-prediction-and-design"
|
||
],
|
||
"tasks": [
|
||
"Foundation Model",
|
||
"Protein Structure Prediction"
|
||
],
|
||
"modalities": [
|
||
"Protein"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "chatdrug",
|
||
"name": "ChatDrug",
|
||
"type": "model",
|
||
"url": "https://github.com/chao1224/ChatDrug",
|
||
"description": "LLM-based conversational pipeline for drug discovery, using natural language prompts for iterative drug editing and optimization.",
|
||
"tags": [
|
||
"llm-for-biology"
|
||
],
|
||
"tasks": [
|
||
"Language Modeling"
|
||
],
|
||
"modalities": [
|
||
"Text"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "chemberta_2",
|
||
"name": "ChemBERTa-2",
|
||
"type": "model",
|
||
"url": "https://github.com/seyonechithrananda/bert-loves-chemistry",
|
||
"description": "RoBERTa-based molecular language model pretrained on SMILES for small-molecule representation learning.",
|
||
"tags": [
|
||
"compound-embedding",
|
||
"compound-foundation-models",
|
||
"foundation-models"
|
||
],
|
||
"tasks": [
|
||
"representation-learning"
|
||
],
|
||
"modalities": [
|
||
"chemical-structure"
|
||
],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"molecule"
|
||
],
|
||
"methods": [
|
||
"language-model",
|
||
"self-supervised-learning",
|
||
"transformer"
|
||
],
|
||
"github": "https://github.com/seyonechithrananda/bert-loves-chemistry",
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://github.com/seyonechithrananda/bert-loves-chemistry"
|
||
]
|
||
},
|
||
{
|
||
"id": "chemcpa",
|
||
"name": "chemCPA",
|
||
"type": "model",
|
||
"url": "https://github.com/theislab/chemCPA",
|
||
"description": "Compositional perturbation autoencoder for predicting single-cell transcriptional responses to unseen drug perturbations and dose combinations.",
|
||
"tags": [
|
||
"drug-discovery",
|
||
"drug-perturbation"
|
||
],
|
||
"tasks": [
|
||
"Drug Discovery",
|
||
"Drug Perturbation"
|
||
],
|
||
"modalities": [
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "chief",
|
||
"name": "CHIEF",
|
||
"type": "model",
|
||
"url": "https://github.com/hms-dbmi/CHIEF",
|
||
"description": "Clinical Histopathology Imaging Evaluation Foundation model integrating histology images and clinical context for pan-cancer analysis.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"multi-modal-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"Foundation Model"
|
||
],
|
||
"modalities": [
|
||
"Multi-Modal"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "clawbio",
|
||
"name": "ClawBio",
|
||
"type": "model",
|
||
"url": "https://github.com/ClawBio/ClawBio",
|
||
"description": "Bioinformatics-native AI agent skill library with local-first pharmacogenomics, ancestry PCA, semantic similarity, nutrigenomics, and metagenomics skills.",
|
||
"tags": [
|
||
"llm-for-biology"
|
||
],
|
||
"tasks": [
|
||
"Language Modeling"
|
||
],
|
||
"modalities": [
|
||
"Text"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "cmonge",
|
||
"name": "CMonge",
|
||
"type": "model",
|
||
"url": "https://github.com/AI4SCR/conditional-monge-gap",
|
||
"description": "Conditional optimal transport model for generalizable single-cell perturbation response prediction across drugs and doses.",
|
||
"tags": [
|
||
"drug-discovery",
|
||
"drug-perturbation"
|
||
],
|
||
"tasks": [
|
||
"Drug Discovery",
|
||
"Drug Perturbation"
|
||
],
|
||
"modalities": [
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "concerto",
|
||
"name": "Concerto",
|
||
"type": "model",
|
||
"url": "https://github.com/melobio/Concerto-reproducibility",
|
||
"description": "Contrastive self-supervised learning framework for single-cell multimodal data integration, batch correction, and reference-query mapping.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"multi-omics-foundation-models",
|
||
"single-cell-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"Foundation Model"
|
||
],
|
||
"modalities": [
|
||
"Multi-Omics",
|
||
"Single Cell"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "conch",
|
||
"name": "CONCH",
|
||
"type": "model",
|
||
"url": "https://github.com/mahmoodlab/CONCH",
|
||
"description": "Vision-language foundation model for computational pathology trained with contrastive captioning on pathology imageโtext pairs.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"single-cell-foundation-models",
|
||
"spatial-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"foundation-model-pretraining",
|
||
"representation-learning"
|
||
],
|
||
"modalities": [
|
||
"histopathology",
|
||
"imaging"
|
||
],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"tissue"
|
||
],
|
||
"methods": [
|
||
"contrastive-learning",
|
||
"transformer"
|
||
],
|
||
"github": "https://github.com/mahmoodlab/CONCH",
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://github.com/mahmoodlab/CONCH"
|
||
]
|
||
},
|
||
{
|
||
"id": "cyclecdr",
|
||
"name": "cycleCDR",
|
||
"type": "model",
|
||
"url": "https://github.com/hliulab/cycleCDR",
|
||
"description": "Interpretable cycle-consistency framework for modeling cellular responses to drug perturbations.",
|
||
"tags": [
|
||
"drug-discovery",
|
||
"drug-perturbation"
|
||
],
|
||
"tasks": [
|
||
"Drug Discovery",
|
||
"Drug Perturbation"
|
||
],
|
||
"modalities": [
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "deepaeg",
|
||
"name": "DeepAEG",
|
||
"type": "model",
|
||
"url": "https://github.com/zhejiangzhuque/DeepAEG",
|
||
"description": "GNN embedding + attention mechanism.",
|
||
"tags": [
|
||
"drug-discovery",
|
||
"drug-response-prediction"
|
||
],
|
||
"tasks": [
|
||
"Drug Discovery",
|
||
"Drug Response Prediction"
|
||
],
|
||
"modalities": [
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "deepdsc",
|
||
"name": "DeepDSC",
|
||
"type": "model",
|
||
"url": "https://ieeexplore-ieee-org.ezp2.lib.umn.edu/stamp/stamp.jsp?tp=&arnumber=8723620&tag=1",
|
||
"description": "Autoencoder + fully connected NN.",
|
||
"tags": [
|
||
"drug-discovery",
|
||
"drug-response-prediction"
|
||
],
|
||
"tasks": [
|
||
"Drug Discovery",
|
||
"Drug Response Prediction"
|
||
],
|
||
"modalities": [
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "deepdta",
|
||
"name": "DeepDTA",
|
||
"type": "model",
|
||
"url": "https://github.com/hkmztrk/DeepDTA",
|
||
"description": "Deep learning model using CNNs on protein sequences and drug SMILES.",
|
||
"tags": [
|
||
"drug-discovery",
|
||
"drug-target-interaction"
|
||
],
|
||
"tasks": [
|
||
"Drug Discovery",
|
||
"Drug Target Interaction"
|
||
],
|
||
"modalities": [
|
||
"Protein",
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "deeppurpose",
|
||
"name": "DeepPurpose",
|
||
"type": "model",
|
||
"url": "https://github.com/kexinhuang12345/DeepPurpose",
|
||
"description": "Deep learning library for drug repurposing.",
|
||
"tags": [
|
||
"drug-discovery",
|
||
"drug-repurposing"
|
||
],
|
||
"tasks": [
|
||
"Drug Discovery",
|
||
"Drug Repurposing"
|
||
],
|
||
"modalities": [
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "deepsea",
|
||
"name": "DeepSEA",
|
||
"type": "model",
|
||
"url": "http://deepsea.princeton.edu/",
|
||
"description": "Deep learning framework for predicting chromatin effects of sequence alterations with single-nucleotide sensitivity across thousands of chromatin features.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"genomics-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"Foundation Model"
|
||
],
|
||
"modalities": [
|
||
"Genomics"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "deepspot",
|
||
"name": "DeepSpot",
|
||
"type": "model",
|
||
"url": "https://github.com/ratschlab/DeepSpot",
|
||
"description": "Deep learning model predicting spatial transcriptomics from H&E images at spot and single-cell resolution.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"single-cell-foundation-models",
|
||
"spatial-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"regression"
|
||
],
|
||
"modalities": [
|
||
"histopathology",
|
||
"spatial-transcriptomics"
|
||
],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"gene",
|
||
"tissue"
|
||
],
|
||
"github": "https://github.com/ratschlab/DeepSpot",
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://github.com/ratschlab/DeepSpot"
|
||
]
|
||
},
|
||
{
|
||
"id": "deepspot_m",
|
||
"name": "DeepSpot-M",
|
||
"type": "model",
|
||
"url": "https://github.com/ratschlab/DeepSpotM",
|
||
"description": "Multimodal foundation model for transcriptome-wide virtual spatial transcriptomics from histology.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"single-cell-foundation-models",
|
||
"spatial-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"foundation-model-pretraining",
|
||
"regression"
|
||
],
|
||
"modalities": [
|
||
"histopathology",
|
||
"spatial-transcriptomics",
|
||
"transcriptomics"
|
||
],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"gene",
|
||
"tissue"
|
||
],
|
||
"github": "https://github.com/ratschlab/DeepSpotM",
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://github.com/ratschlab/DeepSpotM"
|
||
]
|
||
},
|
||
{
|
||
"id": "deepspot2cell",
|
||
"name": "DeepSpot2Cell",
|
||
"type": "model",
|
||
"url": "https://github.com/ratschlab/DeepSpot2Cell",
|
||
"description": "Predicts virtual single-cell spatial transcriptomics from H&E using spot-level supervision (NeurIPS 2025 Imageomics).",
|
||
"tags": [
|
||
"foundation-models",
|
||
"single-cell-foundation-models",
|
||
"spatial-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"regression"
|
||
],
|
||
"modalities": [
|
||
"histopathology",
|
||
"spatial-transcriptomics"
|
||
],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"cell",
|
||
"gene",
|
||
"tissue"
|
||
],
|
||
"github": "https://github.com/ratschlab/DeepSpot2Cell",
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://github.com/ratschlab/DeepSpot2Cell"
|
||
]
|
||
},
|
||
{
|
||
"id": "dgdrp",
|
||
"name": "DGDRP",
|
||
"type": "model",
|
||
"url": "https://github.com/minwoopak/heteronet",
|
||
"description": "Multi-view embedding neural network.",
|
||
"tags": [
|
||
"drug-discovery",
|
||
"drug-response-prediction"
|
||
],
|
||
"tasks": [
|
||
"Drug Discovery",
|
||
"Drug Response Prediction"
|
||
],
|
||
"modalities": [
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "diffdock",
|
||
"name": "DiffDock",
|
||
"type": "model",
|
||
"url": "https://github.com/gcorso/DiffDock",
|
||
"description": "Diffusion generative model for molecular docking, predicting the binding pose of small molecules to protein targets.",
|
||
"tags": [
|
||
"drug-discovery",
|
||
"molecular-generation"
|
||
],
|
||
"tasks": [
|
||
"docking"
|
||
],
|
||
"modalities": [
|
||
"molecular-structure"
|
||
],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"molecule",
|
||
"protein"
|
||
],
|
||
"methods": [
|
||
"diffusion",
|
||
"geometric-deep-learning"
|
||
],
|
||
"year": 2023,
|
||
"github": "https://github.com/gcorso/DiffDock",
|
||
"paper": "https://openreview.net/forum?id=kKF8_K-mBbS",
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://github.com/gcorso/DiffDock",
|
||
"https://openreview.net/forum?id=kKF8_K-mBbS"
|
||
]
|
||
},
|
||
{
|
||
"id": "diffsbdd",
|
||
"name": "DiffSBDD",
|
||
"type": "model",
|
||
"url": "https://github.com/arneschneuing/DiffSBDD",
|
||
"description": "Equivariant diffusion model for structure-based drug design that generates molecules and binding conformations for protein targets.",
|
||
"tags": [
|
||
"drug-discovery",
|
||
"molecular-generation"
|
||
],
|
||
"tasks": [
|
||
"Drug Discovery",
|
||
"Molecular Generation"
|
||
],
|
||
"modalities": [
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "dnabert",
|
||
"name": "DNABERT",
|
||
"type": "model",
|
||
"url": "https://github.com/jerryji1993/DNABERT",
|
||
"description": "Pre-trained bidirectional encoder for DNA sequence analysis.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"genomics-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"Foundation Model"
|
||
],
|
||
"modalities": [
|
||
"Genomics"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "dnabert_2",
|
||
"name": "DNABERT-2",
|
||
"type": "model",
|
||
"url": "https://github.com/Zhihan1996/DNABERT_2",
|
||
"description": "Improved genome foundation model with efficient tokenization.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"genomics-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"Foundation Model"
|
||
],
|
||
"modalities": [
|
||
"Genomics"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "drgat",
|
||
"name": "drGAT",
|
||
"type": "model",
|
||
"url": "https://github.com/inoue0426/drGAT",
|
||
"description": "Attention-based model for drug response prediction with gene explainability.",
|
||
"tags": [
|
||
"drug-discovery",
|
||
"drug-response-prediction"
|
||
],
|
||
"tasks": [
|
||
"Drug Discovery",
|
||
"Drug Response Prediction"
|
||
],
|
||
"modalities": [
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "drugban",
|
||
"name": "DrugBAN",
|
||
"type": "model",
|
||
"url": "https://github.com/peizhenbai/DrugBAN",
|
||
"description": "Bilinear attention network for interpretable DTI prediction.",
|
||
"tags": [
|
||
"drug-discovery",
|
||
"drug-target-interaction"
|
||
],
|
||
"tasks": [
|
||
"Drug Discovery",
|
||
"Drug Target Interaction"
|
||
],
|
||
"modalities": [
|
||
"Protein",
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "druml",
|
||
"name": "DRUML",
|
||
"type": "model",
|
||
"url": "https://github.com/CutillasLab/DRUMLR",
|
||
"description": "Ensemble machine learning framework combining standard ML with deep learning to systematically rank anti-cancer drugs from proteomics and RNA-seq data.",
|
||
"tags": [
|
||
"drug-discovery",
|
||
"drug-response-prediction"
|
||
],
|
||
"tasks": [
|
||
"Drug Discovery",
|
||
"Drug Response Prediction"
|
||
],
|
||
"modalities": [
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "dtinet",
|
||
"name": "DTINet",
|
||
"type": "model",
|
||
"url": "https://github.com/luoyunan/DTINet",
|
||
"description": "Network-based framework integrating heterogeneous biological data for DTI prediction.",
|
||
"tags": [
|
||
"drug-discovery",
|
||
"drug-target-interaction"
|
||
],
|
||
"tasks": [
|
||
"Drug Discovery",
|
||
"Drug Target Interaction"
|
||
],
|
||
"modalities": [
|
||
"Protein",
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "enformer",
|
||
"name": "Enformer",
|
||
"type": "model",
|
||
"url": "https://github.com/deepmind/deepmind-research/tree/master/enformer",
|
||
"description": "Transformer model predicting gene expression from DNA sequence.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"genomics-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"Foundation Model"
|
||
],
|
||
"modalities": [
|
||
"Genomics"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "esm3",
|
||
"name": "ESM3",
|
||
"type": "model",
|
||
"url": "https://github.com/evolutionaryscale/esm",
|
||
"description": "Multimodal protein language model that jointly reasons over sequence, structure, and function for generative protein design and engineering.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"protein-foundation-models",
|
||
"protein-structure-prediction-and-design"
|
||
],
|
||
"tasks": [
|
||
"protein-sequence-design",
|
||
"representation-learning"
|
||
],
|
||
"modalities": [
|
||
"molecular-structure",
|
||
"protein-sequence"
|
||
],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"protein"
|
||
],
|
||
"methods": [
|
||
"generative-model",
|
||
"language-model",
|
||
"transformer"
|
||
],
|
||
"github": "https://github.com/evolutionaryscale/esm",
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://github.com/evolutionaryscale/esm"
|
||
]
|
||
},
|
||
{
|
||
"id": "esmfold",
|
||
"name": "ESMFold",
|
||
"type": "model",
|
||
"url": "https://github.com/facebookresearch/esm",
|
||
"description": "Fast protein structure prediction using language model embeddings.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"protein-foundation-models",
|
||
"protein-structure-prediction-and-design"
|
||
],
|
||
"tasks": [
|
||
"representation-learning",
|
||
"structure-prediction"
|
||
],
|
||
"modalities": [
|
||
"molecular-structure",
|
||
"protein-sequence"
|
||
],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"protein"
|
||
],
|
||
"methods": [
|
||
"language-model",
|
||
"transformer"
|
||
],
|
||
"year": 2023,
|
||
"github": "https://github.com/facebookresearch/esm",
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://github.com/facebookresearch/esm"
|
||
]
|
||
},
|
||
{
|
||
"id": "evo",
|
||
"name": "Evo",
|
||
"type": "model",
|
||
"url": "https://github.com/evo-design/evo",
|
||
"description": "Long-context genomic foundation model (up to 1M tokens).",
|
||
"tags": [
|
||
"foundation-models",
|
||
"genomics-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"Foundation Model"
|
||
],
|
||
"modalities": [
|
||
"Genomics"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "evodiff",
|
||
"name": "EvoDiff",
|
||
"type": "model",
|
||
"url": "https://github.com/microsoft/evodiff",
|
||
"description": "Discrete diffusion framework for protein sequence generation trained on evolutionary-scale data, supporting unconditional generation, disordered region design, and functional motif scaffolding. [ [paper-2023](https://www.biorxiv.org/content/10.1101/2023.09.11.556673v1) ]",
|
||
"tags": [
|
||
"foundation-models",
|
||
"protein-foundation-models",
|
||
"protein-structure-prediction-and-design"
|
||
],
|
||
"tasks": [
|
||
"Foundation Model",
|
||
"Protein Structure Prediction"
|
||
],
|
||
"modalities": [
|
||
"Protein"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "evolutionary_scale_modeling_esm",
|
||
"name": "Evolutionary Scale Modeling (ESM)",
|
||
"type": "model",
|
||
"url": "https://github.com/facebookresearch/esm",
|
||
"description": "Protein embeddings.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"pre-trained-embedding",
|
||
"protein-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"representation-learning"
|
||
],
|
||
"modalities": [
|
||
"protein-sequence"
|
||
],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"protein"
|
||
],
|
||
"methods": [
|
||
"language-model",
|
||
"self-supervised-learning",
|
||
"transformer"
|
||
],
|
||
"github": "https://github.com/facebookresearch/esm",
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://github.com/facebookresearch/esm"
|
||
]
|
||
},
|
||
{
|
||
"id": "gears",
|
||
"name": "GEARS",
|
||
"type": "model",
|
||
"url": "https://github.com/snap-stanford/GEARS",
|
||
"description": "Graph-based model for predicting transcriptional responses to single and combinatorial genetic perturbations using biological priors.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"single-cell-foundation-models",
|
||
"transcriptomics-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"Foundation Model"
|
||
],
|
||
"modalities": [
|
||
"Single Cell",
|
||
"Transcriptomics"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "genecompass",
|
||
"name": "GeneCompass",
|
||
"type": "model",
|
||
"url": "https://github.com/xCompass-AI/GeneCompass",
|
||
"description": "Large-scale foundation model integrating DNA regulatory sequences and single-cell transcriptomics from 120M+ cells across multiple species for gene regulation prediction.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"multi-omics-foundation-models",
|
||
"single-cell-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"foundation-model-pretraining",
|
||
"representation-learning"
|
||
],
|
||
"modalities": [
|
||
"single-cell-rna-seq",
|
||
"transcriptomics"
|
||
],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"cell",
|
||
"gene"
|
||
],
|
||
"methods": [
|
||
"self-supervised-learning",
|
||
"transformer"
|
||
],
|
||
"year": 2024,
|
||
"github": "https://github.com/xCompass-AI/GeneCompass",
|
||
"paper": "https://www.nature.com/articles/s41422-024-01034-y",
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://github.com/xCompass-AI/GeneCompass",
|
||
"https://www.nature.com/articles/s41422-024-01034-y"
|
||
]
|
||
},
|
||
{
|
||
"id": "geneformer",
|
||
"name": "Geneformer",
|
||
"type": "model",
|
||
"url": "https://huggingface.co/ctheodoris/Geneformer",
|
||
"description": "Context-aware, attention-based deep learning model pretrained on a large corpus of single-cell transcriptomes.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"single-cell-foundation-models",
|
||
"transcriptomics-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"classification",
|
||
"foundation-model-pretraining",
|
||
"perturbation-prediction",
|
||
"representation-learning"
|
||
],
|
||
"modalities": [
|
||
"single-cell-rna-seq",
|
||
"transcriptomics"
|
||
],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"cell",
|
||
"gene"
|
||
],
|
||
"methods": [
|
||
"self-supervised-learning",
|
||
"transformer"
|
||
],
|
||
"year": 2023,
|
||
"documentation": "https://geneformer.readthedocs.io/",
|
||
"paper": "https://www.nature.com/articles/s41586-023-06139-9",
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://huggingface.co/ctheodoris/Geneformer",
|
||
"https://www.nature.com/articles/s41586-023-06139-9"
|
||
]
|
||
},
|
||
{
|
||
"id": "genegpt",
|
||
"name": "GeneGPT",
|
||
"type": "model",
|
||
"url": "https://github.com/ncbi/GeneGPT",
|
||
"description": "LLM for biomedical information, integrated with various APIs.",
|
||
"tags": [
|
||
"llm-for-biology"
|
||
],
|
||
"tasks": [
|
||
"Language Modeling"
|
||
],
|
||
"modalities": [
|
||
"Text"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "genept",
|
||
"name": "GenePT",
|
||
"type": "model",
|
||
"url": "https://github.com/yiqunchen/GenePT",
|
||
"description": "Foundation LLM for single-cell data.",
|
||
"tags": [
|
||
"llm-for-biology"
|
||
],
|
||
"tasks": [
|
||
"batch-correction",
|
||
"classification",
|
||
"representation-learning"
|
||
],
|
||
"modalities": [
|
||
"single-cell-rna-seq",
|
||
"transcriptomics"
|
||
],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"cell",
|
||
"gene"
|
||
],
|
||
"methods": [
|
||
"language-model"
|
||
],
|
||
"year": 2023,
|
||
"github": "https://github.com/yiqunchen/GenePT",
|
||
"paper": "https://www.biorxiv.org/content/10.1101/2023.10.16.562533v2",
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://github.com/yiqunchen/GenePT",
|
||
"https://www.biorxiv.org/content/10.1101/2023.10.16.562533v2"
|
||
]
|
||
},
|
||
{
|
||
"id": "gigapath",
|
||
"name": "GigaPath",
|
||
"type": "model",
|
||
"url": "https://github.com/prov-gigapath/prov-gigapath",
|
||
"description": "Slide-level digital pathology foundation model pretrained on 1.3 billion pathology image tokens from whole-slide images.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"single-cell-foundation-models",
|
||
"spatial-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"foundation-model-pretraining",
|
||
"representation-learning"
|
||
],
|
||
"modalities": [
|
||
"histopathology",
|
||
"imaging"
|
||
],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"tissue"
|
||
],
|
||
"methods": [
|
||
"self-supervised-learning",
|
||
"transformer"
|
||
],
|
||
"github": "https://github.com/prov-gigapath/prov-gigapath",
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://github.com/prov-gigapath/prov-gigapath"
|
||
]
|
||
},
|
||
{
|
||
"id": "glue",
|
||
"name": "GLUE",
|
||
"type": "model",
|
||
"url": "https://github.com/gao-lab/GLUE",
|
||
"description": "Graph-Linked Unified Embedding framework for unpaired single-cell multi-omics data integration across RNA, ATAC, methylation, and protein modalities.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"multi-omics-foundation-models",
|
||
"single-cell-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"Foundation Model"
|
||
],
|
||
"modalities": [
|
||
"Multi-Omics",
|
||
"Single Cell"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "gpn_genomic_pre_trained_network",
|
||
"name": "GPN (Genomic Pre-trained Network)",
|
||
"type": "model",
|
||
"url": "https://github.com/songlab-cal/gpn",
|
||
"description": "Masked language model for DNA sequences enabling zero-shot variant effect prediction without requiring functional annotations.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"genomics-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"Foundation Model"
|
||
],
|
||
"modalities": [
|
||
"Genomics"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "graphdta",
|
||
"name": "GraphDTA",
|
||
"type": "model",
|
||
"url": "https://github.com/thinng/GraphDTA",
|
||
"description": "Graph neural networkโbased DTI prediction using molecular graphs.",
|
||
"tags": [
|
||
"drug-discovery",
|
||
"drug-target-interaction"
|
||
],
|
||
"tasks": [
|
||
"Drug Discovery",
|
||
"Drug Target Interaction"
|
||
],
|
||
"modalities": [
|
||
"Protein",
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "grover",
|
||
"name": "GROVER",
|
||
"type": "model",
|
||
"url": "https://github.com/tencent-ailab/grover",
|
||
"description": "Self-supervised graph transformer for large-scale molecular representation learning from unlabeled compounds.",
|
||
"tags": [
|
||
"compound-embedding",
|
||
"compound-foundation-models",
|
||
"foundation-models"
|
||
],
|
||
"tasks": [
|
||
"representation-learning"
|
||
],
|
||
"modalities": [
|
||
"chemical-structure"
|
||
],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"molecule"
|
||
],
|
||
"methods": [
|
||
"graph-neural-network",
|
||
"self-supervised-learning",
|
||
"transformer"
|
||
],
|
||
"github": "https://github.com/tencent-ailab/grover",
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://github.com/tencent-ailab/grover"
|
||
]
|
||
},
|
||
{
|
||
"id": "hidra",
|
||
"name": "HiDRA",
|
||
"type": "model",
|
||
"url": "https://github.com/bsml320/HiDRA",
|
||
"description": "Hierarchical network model incorporating gene and pathway-level information for cancer drug response prediction.",
|
||
"tags": [
|
||
"drug-discovery",
|
||
"drug-response-prediction"
|
||
],
|
||
"tasks": [
|
||
"Drug Discovery",
|
||
"Drug Response Prediction"
|
||
],
|
||
"modalities": [
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "hyenadna",
|
||
"name": "HyenaDNA",
|
||
"type": "model",
|
||
"url": "https://github.com/HazyResearch/hyena-dna",
|
||
"description": "Long-range genomic foundation model handling sequences up to 1M tokens with sub-quadratic attention.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"genomics-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"Foundation Model"
|
||
],
|
||
"modalities": [
|
||
"Genomics"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "jamie",
|
||
"name": "JAMIE",
|
||
"type": "model",
|
||
"url": "https://github.com/Oafish1/JAMIE",
|
||
"description": "Joint variational autoencoder for multimodal single-cell data imputation and embedding.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"multi-omics-foundation-models",
|
||
"single-cell-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"Foundation Model"
|
||
],
|
||
"modalities": [
|
||
"Multi-Omics",
|
||
"Single Cell"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "jtvae",
|
||
"name": "JTVAE",
|
||
"type": "model",
|
||
"url": "https://github.com/wengong-jin/icml18-jtnn",
|
||
"description": "Junction tree variational autoencoder for molecular graph generation that guarantees chemical validity via a hierarchical tree decomposition.",
|
||
"tags": [
|
||
"drug-discovery",
|
||
"molecular-generation"
|
||
],
|
||
"tasks": [
|
||
"Drug Discovery",
|
||
"Molecular Generation"
|
||
],
|
||
"modalities": [
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "matcha",
|
||
"name": "Matcha",
|
||
"type": "model",
|
||
"url": "https://github.com/LigandPro/Matcha",
|
||
"description": "Multi-stage Riemannian flow matching model for physically valid molecular docking with scoring, pose filtering, and benchmarks.",
|
||
"tags": [
|
||
"drug-discovery",
|
||
"molecular-generation"
|
||
],
|
||
"tasks": [
|
||
"Drug Discovery",
|
||
"Molecular Generation"
|
||
],
|
||
"modalities": [
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "mcpinn",
|
||
"name": "MCPINN",
|
||
"type": "model",
|
||
"url": "https://github.com/mhlee0903/multi_channels_PINN",
|
||
"description": "Drug discovery via compound-protein interaction and machine learning.",
|
||
"tags": [
|
||
"compound-protein-interaction",
|
||
"drug-discovery"
|
||
],
|
||
"tasks": [
|
||
"Compound-Protein Interaction",
|
||
"Drug Discovery"
|
||
],
|
||
"modalities": [
|
||
"Protein",
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "midas",
|
||
"name": "MIDAS",
|
||
"type": "model",
|
||
"url": "https://github.com/labomics/midas",
|
||
"description": "Mosaic integration and differential accessibility model for single-cell multi-omics data that handles arbitrary missing-modality combinations across transcriptomics, chromatin accessibility, and proteomics.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"multi-omics-foundation-models",
|
||
"single-cell-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"Foundation Model"
|
||
],
|
||
"modalities": [
|
||
"Multi-Omics",
|
||
"Single Cell"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "mira",
|
||
"name": "MIRA",
|
||
"type": "model",
|
||
"url": "https://github.com/cistrome/MIRA",
|
||
"description": "Probabilistic multimodal topic model jointly modeling single-cell transcriptomics and chromatin accessibility for regulatory network inference.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"multi-omics-foundation-models",
|
||
"single-cell-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"Foundation Model"
|
||
],
|
||
"modalities": [
|
||
"Multi-Omics",
|
||
"Single Cell"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "mofa",
|
||
"name": "MOFA+",
|
||
"type": "model",
|
||
"url": "https://github.com/bioFAM/MOFA2",
|
||
"description": "Multi-Omics Factor Analysis framework identifying shared axes of variation across bulk and single-cell datasets including RNA, ATAC, proteomics, methylation, and copy number.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"multi-omics-foundation-models",
|
||
"single-cell-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"Foundation Model"
|
||
],
|
||
"modalities": [
|
||
"Multi-Omics",
|
||
"Single Cell"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "mofgcn",
|
||
"name": "MOFGCN",
|
||
"type": "model",
|
||
"url": "https://github.com/weiba/MOFGCN/tree/main",
|
||
"description": "GCN + heterogeneous network.",
|
||
"tags": [
|
||
"drug-discovery",
|
||
"drug-response-prediction"
|
||
],
|
||
"tasks": [
|
||
"Drug Discovery",
|
||
"Drug Response Prediction"
|
||
],
|
||
"modalities": [
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "mol2vec",
|
||
"name": "Mol2Vec",
|
||
"type": "model",
|
||
"url": "https://github.com/samoturk/mol2vec",
|
||
"description": "Unsupervised molecular embedding method inspired by Word2Vec for learning vector representations of chemical substructures.",
|
||
"tags": [
|
||
"compound-embedding",
|
||
"compound-foundation-models",
|
||
"foundation-models"
|
||
],
|
||
"tasks": [
|
||
"Foundation Model"
|
||
],
|
||
"modalities": [
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "molecular_transformer",
|
||
"name": "Molecular Transformer",
|
||
"type": "model",
|
||
"url": "https://github.com/pschwllr/MolecularTransformer",
|
||
"description": "Sequence-to-sequence model for retrosynthesis prediction.",
|
||
"tags": [
|
||
"drug-discovery",
|
||
"molecular-generation"
|
||
],
|
||
"tasks": [
|
||
"Drug Discovery",
|
||
"Molecular Generation"
|
||
],
|
||
"modalities": [
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "molformer",
|
||
"name": "MolFormer",
|
||
"type": "model",
|
||
"url": "https://github.com/IBM/molformer",
|
||
"description": "Linear attention transformer pretrained on millions of SMILES strings for efficient molecular embeddings.",
|
||
"tags": [
|
||
"compound-embedding",
|
||
"compound-foundation-models",
|
||
"foundation-models"
|
||
],
|
||
"tasks": [
|
||
"representation-learning"
|
||
],
|
||
"modalities": [
|
||
"chemical-structure"
|
||
],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"molecule"
|
||
],
|
||
"methods": [
|
||
"language-model",
|
||
"self-supervised-learning",
|
||
"transformer"
|
||
],
|
||
"github": "https://github.com/IBM/molformer",
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://github.com/IBM/molformer"
|
||
]
|
||
},
|
||
{
|
||
"id": "molgpt",
|
||
"name": "MolGPT",
|
||
"type": "model",
|
||
"url": "https://github.com/devalab/molgpt",
|
||
"description": "Transformer-based model for molecular generation.",
|
||
"tags": [
|
||
"drug-discovery",
|
||
"molecular-generation"
|
||
],
|
||
"tasks": [
|
||
"Drug Discovery",
|
||
"Molecular Generation"
|
||
],
|
||
"modalities": [
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "molt5",
|
||
"name": "MolT5",
|
||
"type": "model",
|
||
"url": "https://github.com/blender-nlp/MolT5",
|
||
"description": "Language model for molecular tasks bridging text and SMILES, enabling molecule captioning and text-driven molecule generation.",
|
||
"tags": [
|
||
"llm-for-biology"
|
||
],
|
||
"tasks": [
|
||
"Language Modeling"
|
||
],
|
||
"modalities": [
|
||
"Text"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "moltrans",
|
||
"name": "MolTrans",
|
||
"type": "model",
|
||
"url": "https://github.com/kexinhuang12345/MolTrans",
|
||
"description": "Transformer-based DTI model leveraging molecular substructures.",
|
||
"tags": [
|
||
"drug-discovery",
|
||
"drug-target-interaction"
|
||
],
|
||
"tasks": [
|
||
"Drug Discovery",
|
||
"Drug Target Interaction"
|
||
],
|
||
"modalities": [
|
||
"Protein",
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "multigrate",
|
||
"name": "Multigrate",
|
||
"type": "model",
|
||
"url": "https://github.com/theislab/multigrate",
|
||
"description": "Asymmetric multi-omics variational autoencoder for integrating single-cell data across RNA, ATAC, and protein modalities with missing-modality support.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"multi-omics-foundation-models",
|
||
"single-cell-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"Foundation Model"
|
||
],
|
||
"modalities": [
|
||
"Multi-Omics",
|
||
"Single Cell"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "multivi",
|
||
"name": "MultiVI",
|
||
"type": "model",
|
||
"url": "https://github.com/scverse/scvi-tools",
|
||
"description": "Multi-modal variational autoencoder for integrating paired and unpaired single-cell RNA-seq and ATAC-seq measurements into a unified latent space.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"multi-omics-foundation-models",
|
||
"single-cell-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"Foundation Model"
|
||
],
|
||
"modalities": [
|
||
"Multi-Omics",
|
||
"Single Cell"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "musk",
|
||
"name": "MUSK",
|
||
"type": "model",
|
||
"url": "https://github.com/lilab-stanford/MUSK",
|
||
"description": "Vision-language foundation model for precision oncology analyzing multimodal paired text and pathology image data for biomarker prediction and retrieval.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"multi-modal-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"Foundation Model"
|
||
],
|
||
"modalities": [
|
||
"Multi-Modal"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "neodti",
|
||
"name": "NeoDTI",
|
||
"type": "model",
|
||
"url": "https://github.com/FangpingWan/NeoDTI",
|
||
"description": "Library for drug-target interaction prediction.",
|
||
"tags": [
|
||
"drug-discovery",
|
||
"drug-target-interaction"
|
||
],
|
||
"tasks": [
|
||
"Drug Discovery",
|
||
"Drug Target Interaction"
|
||
],
|
||
"modalities": [
|
||
"Protein",
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "nicheformer",
|
||
"name": "Nicheformer",
|
||
"type": "model",
|
||
"url": "https://github.com/theislab/nicheformer",
|
||
"description": "Foundation model for single-cell and spatial omics using a transformer architecture with positional embeddings to encode spatial cell information.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"single-cell-foundation-models",
|
||
"spatial-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"foundation-model-pretraining",
|
||
"representation-learning"
|
||
],
|
||
"modalities": [
|
||
"single-cell-rna-seq",
|
||
"spatial-transcriptomics",
|
||
"transcriptomics"
|
||
],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"cell",
|
||
"gene",
|
||
"tissue"
|
||
],
|
||
"methods": [
|
||
"self-supervised-learning",
|
||
"transformer"
|
||
],
|
||
"year": 2024,
|
||
"github": "https://github.com/theislab/nicheformer",
|
||
"paper": "https://doi.org/10.1101/2024.04.15.589472",
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://github.com/theislab/nicheformer",
|
||
"https://doi.org/10.1101/2024.04.15.589472"
|
||
]
|
||
},
|
||
{
|
||
"id": "nucleotide_transformer",
|
||
"name": "Nucleotide Transformer",
|
||
"type": "model",
|
||
"url": "https://github.com/instadeepai/nucleotide-transformer",
|
||
"description": "Foundation model for genomic sequences across multiple species.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"genomics-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"Foundation Model"
|
||
],
|
||
"modalities": [
|
||
"Genomics"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "omegafold",
|
||
"name": "OmegaFold",
|
||
"type": "model",
|
||
"url": "https://github.com/HeliXonProtein/OmegaFold",
|
||
"description": "High-resolution de novo protein structure prediction from sequence.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"protein-foundation-models",
|
||
"protein-structure-prediction-and-design"
|
||
],
|
||
"tasks": [
|
||
"Foundation Model",
|
||
"Protein Structure Prediction"
|
||
],
|
||
"modalities": [
|
||
"Protein"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "openfold",
|
||
"name": "OpenFold",
|
||
"type": "model",
|
||
"url": "https://github.com/aqlaboratory/openfold",
|
||
"description": "Trainable, memory-efficient open-source reproduction of AlphaFold2 enabling custom protein structure prediction workflows.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"protein-foundation-models",
|
||
"protein-structure-prediction-and-design"
|
||
],
|
||
"tasks": [
|
||
"Foundation Model",
|
||
"Protein Structure Prediction"
|
||
],
|
||
"modalities": [
|
||
"Protein"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "paccmannrl",
|
||
"name": "PaccMannRL",
|
||
"type": "model",
|
||
"url": "https://github.com/PaccMann/paccmann_generator",
|
||
"description": "Reinforcement learning-based generative model for de novo hit-like anticancer molecule design from transcriptomic data.",
|
||
"tags": [
|
||
"drug-discovery",
|
||
"molecular-generation"
|
||
],
|
||
"tasks": [
|
||
"Drug Discovery",
|
||
"Molecular Generation"
|
||
],
|
||
"modalities": [
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "pathomicfusion",
|
||
"name": "PathomicFusion",
|
||
"type": "model",
|
||
"url": "https://github.com/mahmoodlab/PathomicFusion",
|
||
"description": "Integrated framework fusing histopathology and genomic features via CNN, GNN, and attention gating for cancer diagnosis and prognosis.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"multi-modal-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"Foundation Model"
|
||
],
|
||
"modalities": [
|
||
"Multi-Modal"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "phikon",
|
||
"name": "Phikon",
|
||
"type": "model",
|
||
"url": "https://huggingface.co/owkin/phikon",
|
||
"description": "ViT-based pathology foundation model pretrained with iBOT self-supervision on TCGA whole-slide images.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"single-cell-foundation-models",
|
||
"spatial-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"foundation-model-pretraining",
|
||
"representation-learning"
|
||
],
|
||
"modalities": [
|
||
"histopathology",
|
||
"imaging"
|
||
],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"tissue"
|
||
],
|
||
"methods": [
|
||
"self-supervised-learning",
|
||
"transformer"
|
||
],
|
||
"documentation": "https://huggingface.co/owkin/phikon",
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://huggingface.co/owkin/phikon"
|
||
]
|
||
},
|
||
{
|
||
"id": "plip",
|
||
"name": "PLIP",
|
||
"type": "model",
|
||
"url": "https://github.com/PathologyFoundation/plip",
|
||
"description": "Vision-language foundation model for pathology trained with contrastive learning on pathology imageโtext pairs for image classification and text-to-image retrieval.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"multi-modal-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"classification",
|
||
"representation-learning"
|
||
],
|
||
"modalities": [
|
||
"histopathology",
|
||
"imaging"
|
||
],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"tissue"
|
||
],
|
||
"methods": [
|
||
"contrastive-learning"
|
||
],
|
||
"github": "https://github.com/PathologyFoundation/plip",
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://github.com/PathologyFoundation/plip"
|
||
]
|
||
},
|
||
{
|
||
"id": "porpoise",
|
||
"name": "PORPOISE",
|
||
"type": "model",
|
||
"url": "https://github.com/mahmoodlab/PORPOISE",
|
||
"description": "Pan-cancer integrative histology-genomic analysis framework using multimodal deep learning for patient stratification.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"multi-modal-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"Foundation Model"
|
||
],
|
||
"modalities": [
|
||
"Multi-Modal"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "prnet",
|
||
"name": "PRNet",
|
||
"type": "model",
|
||
"url": "https://github.com/Perturbation-Response-Prediction/PRnet",
|
||
"description": "Deep generative model for predicting transcriptional responses to novel chemical perturbations for drug discovery.",
|
||
"tags": [
|
||
"drug-discovery",
|
||
"drug-perturbation"
|
||
],
|
||
"tasks": [
|
||
"Drug Discovery",
|
||
"Drug Perturbation"
|
||
],
|
||
"modalities": [
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "progen2",
|
||
"name": "ProGen2",
|
||
"type": "model",
|
||
"url": "https://github.com/salesforce/progen",
|
||
"description": "Protein language model trained on diverse protein families for sequence generation and fitness prediction.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"pre-trained-embedding",
|
||
"protein-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"protein-sequence-design",
|
||
"representation-learning"
|
||
],
|
||
"modalities": [
|
||
"protein-sequence"
|
||
],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"protein"
|
||
],
|
||
"methods": [
|
||
"generative-model",
|
||
"language-model",
|
||
"transformer"
|
||
],
|
||
"github": "https://github.com/salesforce/progen",
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://github.com/salesforce/progen"
|
||
]
|
||
},
|
||
{
|
||
"id": "proteinmpnn",
|
||
"name": "ProteinMPNN",
|
||
"type": "model",
|
||
"url": "https://github.com/dauparas/ProteinMPNN",
|
||
"description": "Deep learning model for protein sequence design given backbone structure.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"protein-foundation-models",
|
||
"protein-structure-prediction-and-design"
|
||
],
|
||
"tasks": [
|
||
"protein-sequence-design"
|
||
],
|
||
"modalities": [
|
||
"molecular-structure",
|
||
"protein-sequence"
|
||
],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"protein"
|
||
],
|
||
"methods": [
|
||
"graph-neural-network",
|
||
"message-passing-neural-network"
|
||
],
|
||
"year": 2022,
|
||
"github": "https://github.com/dauparas/ProteinMPNN",
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://github.com/dauparas/ProteinMPNN"
|
||
]
|
||
},
|
||
{
|
||
"id": "prottrans",
|
||
"name": "ProtTrans",
|
||
"type": "model",
|
||
"url": "https://github.com/agemagician/ProtTrans",
|
||
"description": "Suite of protein language models (ProtBERT, ProtT5, ProtXLNet) trained on billions of protein sequences from UniRef and BFD.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"pre-trained-embedding",
|
||
"protein-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"representation-learning"
|
||
],
|
||
"modalities": [
|
||
"protein-sequence"
|
||
],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"protein"
|
||
],
|
||
"methods": [
|
||
"language-model",
|
||
"self-supervised-learning",
|
||
"transformer"
|
||
],
|
||
"github": "https://github.com/agemagician/ProtTrans",
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://github.com/agemagician/ProtTrans"
|
||
]
|
||
},
|
||
{
|
||
"id": "recover",
|
||
"name": "RECOVER",
|
||
"type": "model",
|
||
"url": "https://github.com/RECOVERcoalition/Recover",
|
||
"description": "Machine learning framework for predicting synergistic drug combination responses across cell lines.",
|
||
"tags": [
|
||
"drug-discovery",
|
||
"drug-response-prediction"
|
||
],
|
||
"tasks": [
|
||
"Drug Discovery",
|
||
"Drug Response Prediction"
|
||
],
|
||
"modalities": [
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "reinvent",
|
||
"name": "REINVENT",
|
||
"type": "model",
|
||
"url": "https://github.com/MolecularAI/Reinvent",
|
||
"description": "Reinforcement learning for de novo drug design.",
|
||
"tags": [
|
||
"drug-discovery",
|
||
"molecular-generation"
|
||
],
|
||
"tasks": [
|
||
"Drug Discovery",
|
||
"Molecular Generation"
|
||
],
|
||
"modalities": [
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "release",
|
||
"name": "ReLeaSE",
|
||
"type": "model",
|
||
"url": "https://github.com/isayev/ReLeaSE",
|
||
"description": "Deep reinforcement learning framework for de novo drug design combining a generative and predictive model.",
|
||
"tags": [
|
||
"drug-discovery",
|
||
"molecular-generation"
|
||
],
|
||
"tasks": [
|
||
"Drug Discovery",
|
||
"Molecular Generation"
|
||
],
|
||
"modalities": [
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "rfdiffusion",
|
||
"name": "RFdiffusion",
|
||
"type": "model",
|
||
"url": "https://github.com/RosettaCommons/RFdiffusion",
|
||
"description": "Generative model for protein backbone design using diffusion.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"protein-foundation-models",
|
||
"protein-structure-prediction-and-design"
|
||
],
|
||
"tasks": [
|
||
"Foundation Model",
|
||
"Protein Structure Prediction"
|
||
],
|
||
"modalities": [
|
||
"Protein"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "rosettafold",
|
||
"name": "RoseTTAFold",
|
||
"type": "model",
|
||
"url": "https://github.com/RosettaCommons/RoseTTAFold",
|
||
"description": "Three-track neural network for protein structure prediction.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"protein-foundation-models",
|
||
"protein-structure-prediction-and-design"
|
||
],
|
||
"tasks": [
|
||
"Foundation Model",
|
||
"Protein Structure Prediction"
|
||
],
|
||
"modalities": [
|
||
"Protein"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "saprot",
|
||
"name": "SaProt",
|
||
"type": "model",
|
||
"url": "https://github.com/westlake-reup/SaProt",
|
||
"description": "Structure-aware protein language model using structure-aware tokens that encode both sequence and backbone geometry for improved function prediction.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"protein-foundation-models",
|
||
"protein-structure-prediction-and-design"
|
||
],
|
||
"tasks": [
|
||
"Foundation Model",
|
||
"Protein Structure Prediction"
|
||
],
|
||
"modalities": [
|
||
"Protein"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "saturn",
|
||
"name": "SATURN",
|
||
"type": "model",
|
||
"url": "https://github.com/snap-stanford/SATURN",
|
||
"description": "Transformer-based model integrating gene expression and protein sequences via a protein language model to learn unified multi-species cell embeddings.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"single-cell-foundation-models",
|
||
"transcriptomics-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"Foundation Model"
|
||
],
|
||
"modalities": [
|
||
"Single Cell",
|
||
"Transcriptomics"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "scarches",
|
||
"name": "scArches",
|
||
"type": "model",
|
||
"url": "https://github.com/theislab/scarches",
|
||
"description": "Transfer learning framework for mapping new single-cell datasets onto pre-trained reference atlases across batches, conditions, and modalities.",
|
||
"tags": [
|
||
"domain-alignment",
|
||
"foundation-models",
|
||
"single-cell-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"Domain Alignment",
|
||
"Foundation Model"
|
||
],
|
||
"modalities": [
|
||
"Single Cell"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "scbert",
|
||
"name": "scBERT",
|
||
"type": "model",
|
||
"url": "https://github.com/TencentAILabHealthcare/scBERT",
|
||
"description": "BERT-based foundation model pretrained on large-scale scRNA-seq data for cell type annotation.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"single-cell-foundation-models",
|
||
"transcriptomics-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"cell-type-annotation",
|
||
"classification",
|
||
"foundation-model-pretraining"
|
||
],
|
||
"modalities": [
|
||
"single-cell-rna-seq",
|
||
"transcriptomics"
|
||
],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"cell",
|
||
"gene"
|
||
],
|
||
"methods": [
|
||
"language-model",
|
||
"self-supervised-learning",
|
||
"transformer"
|
||
],
|
||
"year": 2022,
|
||
"github": "https://github.com/TencentAILabHealthcare/scBERT",
|
||
"paper": "https://www.nature.com/articles/s42256-022-00534-z",
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://github.com/TencentAILabHealthcare/scBERT",
|
||
"https://www.nature.com/articles/s42256-022-00534-z"
|
||
]
|
||
},
|
||
{
|
||
"id": "scbutterfly",
|
||
"name": "scButterfly",
|
||
"type": "model",
|
||
"url": "https://github.com/BioX-NKU/scButterfly",
|
||
"description": "Dual-aligned variational autoencoder for single-cell cross-modality translation between paired and unpaired multiomics data.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"multi-omics-foundation-models",
|
||
"single-cell-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"Foundation Model"
|
||
],
|
||
"modalities": [
|
||
"Multi-Omics",
|
||
"Single Cell"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "scfoundation",
|
||
"name": "scFoundation",
|
||
"type": "model",
|
||
"url": "https://github.com/biomap-research/scFoundation",
|
||
"description": "Large-scale foundation model for single-cell gene expression, enabling multiple downstream tasks.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"single-cell-foundation-models",
|
||
"transcriptomics-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"cell-type-annotation",
|
||
"drug-response-prediction",
|
||
"foundation-model-pretraining",
|
||
"perturbation-prediction",
|
||
"representation-learning"
|
||
],
|
||
"modalities": [
|
||
"single-cell-rna-seq",
|
||
"transcriptomics"
|
||
],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"cell",
|
||
"gene"
|
||
],
|
||
"methods": [
|
||
"self-supervised-learning",
|
||
"transformer"
|
||
],
|
||
"year": 2024,
|
||
"github": "https://github.com/biomap-research/scFoundation",
|
||
"paper": "https://www.nature.com/articles/s41592-024-02305-7",
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://github.com/biomap-research/scFoundation",
|
||
"https://www.nature.com/articles/s41592-024-02305-7"
|
||
]
|
||
},
|
||
{
|
||
"id": "scgpt",
|
||
"name": "scGPT",
|
||
"type": "model",
|
||
"url": "https://github.com/bowang-lab/scGPT",
|
||
"description": "Transformer-based foundation model pretrained on millions of single-cell profiles.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"single-cell-foundation-models",
|
||
"transcriptomics-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"cell-type-annotation",
|
||
"foundation-model-pretraining",
|
||
"gene-regulatory-network-inference",
|
||
"perturbation-prediction",
|
||
"representation-learning"
|
||
],
|
||
"modalities": [
|
||
"multi-omics",
|
||
"single-cell-rna-seq",
|
||
"transcriptomics"
|
||
],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"cell",
|
||
"gene"
|
||
],
|
||
"methods": [
|
||
"generative-model",
|
||
"self-supervised-learning",
|
||
"transformer"
|
||
],
|
||
"year": 2024,
|
||
"github": "https://github.com/bowang-lab/scGPT",
|
||
"documentation": "https://scgpt.readthedocs.io/en/latest/",
|
||
"paper": "https://www.nature.com/articles/s41592-024-02201-0",
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://github.com/bowang-lab/scGPT",
|
||
"https://www.nature.com/articles/s41592-024-02201-0"
|
||
]
|
||
},
|
||
{
|
||
"id": "scgpt_spatial",
|
||
"name": "scGPT-spatial",
|
||
"type": "model",
|
||
"url": "https://github.com/bowang-lab/scGPT-spatial",
|
||
"description": "Extension of scGPT for spatial transcriptomics with continual pretraining and a mixture-of-experts decoder for spatial gene expression analysis.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"single-cell-foundation-models",
|
||
"spatial-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"foundation-model-pretraining",
|
||
"imputation",
|
||
"representation-learning"
|
||
],
|
||
"modalities": [
|
||
"multi-omics",
|
||
"single-cell-rna-seq",
|
||
"spatial-transcriptomics"
|
||
],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"cell",
|
||
"gene",
|
||
"tissue"
|
||
],
|
||
"methods": [
|
||
"generative-model",
|
||
"self-supervised-learning",
|
||
"transformer"
|
||
],
|
||
"year": 2025,
|
||
"github": "https://github.com/bowang-lab/scGPT-spatial",
|
||
"paper": "https://www.biorxiv.org/content/10.1101/2025.02.05.636714v1",
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://github.com/bowang-lab/scGPT-spatial",
|
||
"https://www.biorxiv.org/content/10.1101/2025.02.05.636714v1"
|
||
]
|
||
},
|
||
{
|
||
"id": "scmulan",
|
||
"name": "scMulan",
|
||
"type": "model",
|
||
"url": "https://github.com/SuperBianC/scMulan",
|
||
"description": "Single-cell multi-omic language model pretrained on ~10M cells spanning transcriptomics, epigenomics, and proteomics for cross-omics transfer tasks.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"multi-omics-foundation-models",
|
||
"single-cell-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"foundation-model-pretraining",
|
||
"representation-learning"
|
||
],
|
||
"modalities": [
|
||
"epigenomics",
|
||
"multi-omics",
|
||
"proteomics",
|
||
"single-cell-rna-seq",
|
||
"transcriptomics"
|
||
],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"cell",
|
||
"gene"
|
||
],
|
||
"methods": [
|
||
"language-model",
|
||
"transformer"
|
||
],
|
||
"github": "https://github.com/SuperBianC/scMulan",
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://github.com/SuperBianC/scMulan"
|
||
]
|
||
},
|
||
{
|
||
"id": "scpair",
|
||
"name": "scPair",
|
||
"type": "model",
|
||
"url": "https://github.com/quon-titative-biology/scPair",
|
||
"description": "Bidirectional feedforward network for single-cell multimodal analysis with cross-modality prediction leveraging single-cell atlases.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"multi-omics-foundation-models",
|
||
"single-cell-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"Foundation Model"
|
||
],
|
||
"modalities": [
|
||
"Multi-Omics",
|
||
"Single Cell"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "scprint",
|
||
"name": "scPRINT",
|
||
"type": "model",
|
||
"url": "https://github.com/cantinilab/scPRINT",
|
||
"description": "Pretrained on 50M cells for scRNA-seq denoising & zero imputation.",
|
||
"tags": [
|
||
"llm-for-biology"
|
||
],
|
||
"tasks": [
|
||
"batch-correction",
|
||
"cell-type-annotation",
|
||
"foundation-model-pretraining",
|
||
"gene-regulatory-network-inference",
|
||
"imputation",
|
||
"representation-learning"
|
||
],
|
||
"modalities": [
|
||
"single-cell-rna-seq",
|
||
"transcriptomics"
|
||
],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"cell",
|
||
"gene"
|
||
],
|
||
"methods": [
|
||
"self-supervised-learning",
|
||
"transformer"
|
||
],
|
||
"year": 2025,
|
||
"github": "https://github.com/cantinilab/scPRINT",
|
||
"documentation": "https://www.jkobject.com/scPRINT/",
|
||
"paper": "https://www.nature.com/articles/s41467-025-58699-1",
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://github.com/cantinilab/scPRINT",
|
||
"https://www.nature.com/articles/s41467-025-58699-1"
|
||
]
|
||
},
|
||
{
|
||
"id": "sei",
|
||
"name": "Sei",
|
||
"type": "model",
|
||
"url": "https://github.com/FunctionLab/sei-framework",
|
||
"description": "Sequence-to-function framework learning a genome-wide regulatory activity code from DNA sequences for variant effect prediction.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"genomics-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"Foundation Model"
|
||
],
|
||
"modalities": [
|
||
"Genomics"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "spatialglue",
|
||
"name": "SpatialGlue",
|
||
"type": "model",
|
||
"url": "https://github.com/zhanglabtools/SpatialGlue",
|
||
"description": "Graph attention network for spatial multi-omics integration jointly embedding spatial transcriptomics with chromatin accessibility or proteomics.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"multi-omics-foundation-models",
|
||
"single-cell-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"Foundation Model"
|
||
],
|
||
"modalities": [
|
||
"Multi-Omics",
|
||
"Single Cell"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "targetdiff",
|
||
"name": "TargetDiff",
|
||
"type": "model",
|
||
"url": "https://github.com/guanjq/targetdiff",
|
||
"description": "3D equivariant diffusion model for structure-based drug design.",
|
||
"tags": [
|
||
"drug-discovery",
|
||
"molecular-generation"
|
||
],
|
||
"tasks": [
|
||
"Drug Discovery",
|
||
"Molecular Generation"
|
||
],
|
||
"modalities": [
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "tgsa",
|
||
"name": "TGSA",
|
||
"type": "model",
|
||
"url": "https://github.com/violet-sto/TGSA",
|
||
"description": "Tumor gene set and attention-based model leveraging biological pathway knowledge for drug response prediction.",
|
||
"tags": [
|
||
"drug-discovery",
|
||
"drug-response-prediction"
|
||
],
|
||
"tasks": [
|
||
"Drug Discovery",
|
||
"Drug Response Prediction"
|
||
],
|
||
"modalities": [
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "toad",
|
||
"name": "TOAD",
|
||
"type": "model",
|
||
"url": "https://github.com/mahmoodlab/TOAD",
|
||
"description": "Tumor Origin Assessment via Deep-learning; weakly-supervised multi-task model predicting cancer primary origin from H&E whole-slide images.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"multi-modal-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"Foundation Model"
|
||
],
|
||
"modalities": [
|
||
"Multi-Modal"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "tosica",
|
||
"name": "TOSICA",
|
||
"type": "model",
|
||
"url": "https://github.com/JackieHanlaopo/TOSICA",
|
||
"description": "Transformer-based framework for one-stop interpretable cell-type annotation supporting cross-dataset and cross-species transfer.",
|
||
"tags": [
|
||
"domain-alignment",
|
||
"foundation-models",
|
||
"single-cell-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"Domain Alignment",
|
||
"Foundation Model"
|
||
],
|
||
"modalities": [
|
||
"Single Cell"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "totalvi",
|
||
"name": "totalVI",
|
||
"type": "model",
|
||
"url": "https://github.com/scverse/scvi-tools",
|
||
"description": "Probabilistic framework for joint analysis of paired scRNA-seq and protein (CITE-seq) data enabling multi-modal cell state representation across single-cell datasets.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"multi-omics-foundation-models",
|
||
"single-cell-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"Foundation Model"
|
||
],
|
||
"modalities": [
|
||
"Multi-Omics",
|
||
"Single Cell"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "transformercpi",
|
||
"name": "TransformerCPI",
|
||
"type": "model",
|
||
"url": "https://github.com/lifanchen-simm/transformerCPI",
|
||
"description": "CPI prediction using Transformer.",
|
||
"tags": [
|
||
"compound-protein-interaction",
|
||
"drug-discovery"
|
||
],
|
||
"tasks": [
|
||
"Compound-Protein Interaction",
|
||
"Drug Discovery"
|
||
],
|
||
"modalities": [
|
||
"Protein",
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "transigen",
|
||
"name": "TranSiGen",
|
||
"type": "model",
|
||
"url": "https://github.com/myzhengSIMM/TranSiGen",
|
||
"description": "Dual-VAE architecture for ligand-based virtual screening, drug response prediction, and drug repurposing using chemical-induced transcriptional profiles.",
|
||
"tags": [
|
||
"drug-discovery",
|
||
"drug-repurposing"
|
||
],
|
||
"tasks": [
|
||
"Drug Discovery",
|
||
"Drug Repurposing"
|
||
],
|
||
"modalities": [
|
||
"Small Molecule"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "uce",
|
||
"name": "UCE",
|
||
"type": "model",
|
||
"url": "https://github.com/snap-stanford/UCE",
|
||
"description": "Universal Cell Embeddings: zero-shot single-cell embedding model trained on 36M cells across species, tissues, and assays without fine-tuning.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"single-cell-foundation-models",
|
||
"transcriptomics-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"foundation-model-pretraining",
|
||
"representation-learning"
|
||
],
|
||
"modalities": [
|
||
"single-cell-rna-seq",
|
||
"transcriptomics"
|
||
],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"cell"
|
||
],
|
||
"methods": [
|
||
"self-supervised-learning"
|
||
],
|
||
"year": 2026,
|
||
"github": "https://github.com/snap-stanford/UCE",
|
||
"paper": "https://www.nature.com/articles/s41586-026-10689-z",
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://github.com/snap-stanford/UCE",
|
||
"https://www.nature.com/articles/s41586-026-10689-z"
|
||
]
|
||
},
|
||
{
|
||
"id": "uni",
|
||
"name": "UNI",
|
||
"type": "model",
|
||
"url": "https://github.com/mahmoodlab/UNI",
|
||
"description": "General-purpose self-supervised pathology foundation model trained on 100K+ whole-slide images for diverse computational pathology tasks.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"single-cell-foundation-models",
|
||
"spatial-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"foundation-model-pretraining",
|
||
"representation-learning"
|
||
],
|
||
"modalities": [
|
||
"histopathology",
|
||
"imaging"
|
||
],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"tissue"
|
||
],
|
||
"methods": [
|
||
"self-supervised-learning",
|
||
"transformer"
|
||
],
|
||
"github": "https://github.com/mahmoodlab/UNI",
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://github.com/mahmoodlab/UNI"
|
||
]
|
||
},
|
||
{
|
||
"id": "uni_mol",
|
||
"name": "Uni-Mol",
|
||
"type": "model",
|
||
"url": "https://github.com/deepmodeling/Uni-Mol",
|
||
"description": "3D molecular pretraining framework for universal representation learning on molecules and protein pockets.",
|
||
"tags": [
|
||
"compound-embedding",
|
||
"compound-foundation-models",
|
||
"foundation-models"
|
||
],
|
||
"tasks": [
|
||
"docking",
|
||
"representation-learning"
|
||
],
|
||
"modalities": [
|
||
"chemical-structure",
|
||
"molecular-structure"
|
||
],
|
||
"organism": [],
|
||
"api": false,
|
||
"entities": [
|
||
"molecule",
|
||
"protein"
|
||
],
|
||
"methods": [
|
||
"self-supervised-learning",
|
||
"transformer"
|
||
],
|
||
"year": 2023,
|
||
"github": "https://github.com/deepmodeling/Uni-Mol",
|
||
"paper": "https://openreview.net/forum?id=6K2RM6wVqKu",
|
||
"last_checked": "2026-08-08",
|
||
"metadata_sources": [
|
||
"https://github.com/deepmodeling/Uni-Mol",
|
||
"https://openreview.net/forum?id=6K2RM6wVqKu"
|
||
]
|
||
},
|
||
{
|
||
"id": "unitednet",
|
||
"name": "UnitedNet",
|
||
"type": "model",
|
||
"url": "https://github.com/LiuLab-Bioelectronics-Harvard/UnitedNet",
|
||
"description": "Interpretable multi-task deep neural network for single-cell multi-omics integration spanning transcriptomics, chromatin accessibility, and proteomics.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"multi-omics-foundation-models",
|
||
"single-cell-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"Foundation Model"
|
||
],
|
||
"modalities": [
|
||
"Multi-Omics",
|
||
"Single Cell"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "virchow",
|
||
"name": "Virchow",
|
||
"type": "model",
|
||
"url": "https://huggingface.co/paige-ai/Virchow",
|
||
"description": "Million-slide digital pathology foundation model using a vision transformer and self-supervised distillation for tile-level pathology image representation.",
|
||
"tags": [
|
||
"foundation-models",
|
||
"multi-modal-foundation-models"
|
||
],
|
||
"tasks": [
|
||
"Foundation Model"
|
||
],
|
||
"modalities": [
|
||
"Multi-Modal"
|
||
],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "autozyme",
|
||
"name": "AutoZyme",
|
||
"type": "toolkit",
|
||
"url": "https://github.com/ElliotXie/autozyme",
|
||
"description": "Autonomous agentic framework that speeds up bioinformatics software (e.g. Scanpy, Seurat) on CPUs while preserving the original results.",
|
||
"tags": [
|
||
"preprocessing-tools"
|
||
],
|
||
"tasks": [
|
||
"Preprocessing"
|
||
],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "biopython",
|
||
"name": "Biopython",
|
||
"type": "toolkit",
|
||
"url": "https://biopython.org/",
|
||
"description": "Collection of Python tools for biological computation including sequence analysis, structure parsing, and database access.",
|
||
"tags": [
|
||
"preprocessing-tools"
|
||
],
|
||
"tasks": [
|
||
"Preprocessing"
|
||
],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "casper",
|
||
"name": "CaSpER",
|
||
"type": "toolkit",
|
||
"url": "https://github.com/akdess/CaSpER",
|
||
"description": "CNV identification and visualization by integrative analysis of single-cell or bulk RNA-seq data.",
|
||
"tags": [
|
||
"preprocessing-tools"
|
||
],
|
||
"tasks": [
|
||
"Preprocessing"
|
||
],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "cellcharter",
|
||
"name": "CellCharter",
|
||
"type": "toolkit",
|
||
"url": "https://github.com/CSOgroup/cellcharter",
|
||
"description": "Identification and characterization of spatial cell niches from spatial transcriptomics using VAEs and Gaussian mixture models.",
|
||
"tags": [
|
||
"preprocessing-tools"
|
||
],
|
||
"tasks": [
|
||
"Preprocessing"
|
||
],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "cellchat",
|
||
"name": "CellChat",
|
||
"type": "toolkit",
|
||
"url": "https://github.com/sqjin/CellChat",
|
||
"description": "Inference and analysis of cell-cell communication ligand-receptor networks from single-cell transcriptomics data.",
|
||
"tags": [
|
||
"preprocessing-tools"
|
||
],
|
||
"tasks": [
|
||
"Preprocessing"
|
||
],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "celltypist",
|
||
"name": "CellTypist",
|
||
"type": "toolkit",
|
||
"url": "https://github.com/Teichlab/celltypist",
|
||
"description": "Automated cell type annotation for scRNA-seq.",
|
||
"tags": [
|
||
"preprocessing-tools"
|
||
],
|
||
"tasks": [
|
||
"Preprocessing"
|
||
],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "chatspatial",
|
||
"name": "ChatSpatial",
|
||
"type": "toolkit",
|
||
"url": "https://github.com/cafferychen777/ChatSpatial",
|
||
"description": "MCP server for spatial transcriptomics analysis via natural language.",
|
||
"tags": [
|
||
"preprocessing-tools"
|
||
],
|
||
"tasks": [
|
||
"Preprocessing"
|
||
],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "chemistry_development_kit",
|
||
"name": "Chemistry Development Kit",
|
||
"type": "toolkit",
|
||
"url": "https://github.com/cdk/cdk",
|
||
"description": "Cheminformatics software & machine learning tools.",
|
||
"tags": [
|
||
"preprocessing-tools"
|
||
],
|
||
"tasks": [
|
||
"Preprocessing"
|
||
],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "commot",
|
||
"name": "COMMOT",
|
||
"type": "toolkit",
|
||
"url": "https://github.com/zcang/COMMOT",
|
||
"description": "Optimal transport-based framework for screening cell-cell communication in spatial transcriptomics.",
|
||
"tags": [
|
||
"preprocessing-tools"
|
||
],
|
||
"tasks": [
|
||
"Preprocessing"
|
||
],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "deepchem",
|
||
"name": "DeepChem",
|
||
"type": "toolkit",
|
||
"url": "https://github.com/deepchem/deepchem",
|
||
"description": "Deep learning library for drug discovery, quantum chemistry, and materials science.",
|
||
"tags": [
|
||
"preprocessing-tools"
|
||
],
|
||
"tasks": [
|
||
"Preprocessing"
|
||
],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "deeptalk",
|
||
"name": "DeepTalk",
|
||
"type": "toolkit",
|
||
"url": "https://github.com/JiangBioLab/DeepTalk",
|
||
"description": "Graph attention network for deciphering cell-cell communication from spatial transcriptomics data.",
|
||
"tags": [
|
||
"preprocessing-tools"
|
||
],
|
||
"tasks": [
|
||
"Preprocessing"
|
||
],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "doubletfinder",
|
||
"name": "DoubletFinder",
|
||
"type": "toolkit",
|
||
"url": "https://github.com/chris-mcginnis-ucsf/DoubletFinder",
|
||
"description": "Machine learning approach for detecting multiplet (doublet) artifacts in single-cell RNA-seq data.",
|
||
"tags": [
|
||
"preprocessing-tools"
|
||
],
|
||
"tasks": [
|
||
"Preprocessing"
|
||
],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "flashdeconv",
|
||
"name": "FlashDeconv",
|
||
"type": "toolkit",
|
||
"url": "https://github.com/cafferychen777/flashdeconv",
|
||
"description": "High-performance spatial transcriptomics deconvolution (~1M spots in ~3 min).",
|
||
"tags": [
|
||
"preprocessing-tools"
|
||
],
|
||
"tasks": [
|
||
"Preprocessing"
|
||
],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "gromacs",
|
||
"name": "GROMACS",
|
||
"type": "toolkit",
|
||
"url": "https://www.gromacs.org/",
|
||
"description": "Molecular dynamics simulation package for biochemical molecules.",
|
||
"tags": [
|
||
"preprocessing-tools"
|
||
],
|
||
"tasks": [
|
||
"Preprocessing"
|
||
],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "harmony",
|
||
"name": "Harmony",
|
||
"type": "toolkit",
|
||
"url": "https://github.com/immunogenomics/harmony",
|
||
"description": "Fast and scalable integration of single-cell data across datasets, conditions, technologies, and species.",
|
||
"tags": [
|
||
"preprocessing-tools"
|
||
],
|
||
"tasks": [
|
||
"Preprocessing"
|
||
],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "kallisto",
|
||
"name": "kallisto",
|
||
"type": "toolkit",
|
||
"url": "https://pachterlab.github.io/kallisto/",
|
||
"description": "Near-optimal RNA-seq quantification using pseudoalignment for fast transcript abundance estimation.",
|
||
"tags": [
|
||
"preprocessing-tools"
|
||
],
|
||
"tasks": [
|
||
"Preprocessing"
|
||
],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "linger",
|
||
"name": "LINGER",
|
||
"type": "toolkit",
|
||
"url": "https://github.com/Durenlab/LINGER",
|
||
"description": "Neural network for gene regulatory network inference from single-cell multiome (RNA+ATAC-seq) data with bulk data pretraining.",
|
||
"tags": [
|
||
"preprocessing-tools"
|
||
],
|
||
"tasks": [
|
||
"Preprocessing"
|
||
],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "mdanalysis",
|
||
"name": "MDAnalysis",
|
||
"type": "toolkit",
|
||
"url": "https://www.mdanalysis.org/",
|
||
"description": "Python library for analyzing and altering molecular dynamics simulation trajectories.",
|
||
"tags": [
|
||
"preprocessing-tools"
|
||
],
|
||
"tasks": [
|
||
"Preprocessing"
|
||
],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "mogonet",
|
||
"name": "MOGONET",
|
||
"type": "toolkit",
|
||
"url": "https://github.com/txWang/MOGONET",
|
||
"description": "Multi-omics graph convolutional network framework for patient classification and biomarker identification.",
|
||
"tags": [
|
||
"preprocessing-tools"
|
||
],
|
||
"tasks": [
|
||
"Preprocessing"
|
||
],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "monocle3",
|
||
"name": "Monocle3",
|
||
"type": "toolkit",
|
||
"url": "https://cole-trapnell-lab.github.io/monocle3/",
|
||
"description": "Single-cell trajectory analysis tool for learning developmental trajectories and ordering cells in pseudotime.",
|
||
"tags": [
|
||
"preprocessing-tools"
|
||
],
|
||
"tasks": [
|
||
"Preprocessing"
|
||
],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "ncem",
|
||
"name": "NCEM",
|
||
"type": "toolkit",
|
||
"url": "https://github.com/theislab/ncem",
|
||
"description": "GNN-based model for learning intercellular communication from spatial graphs of cells.",
|
||
"tags": [
|
||
"preprocessing-tools"
|
||
],
|
||
"tasks": [
|
||
"Preprocessing"
|
||
],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "numbat",
|
||
"name": "Numbat",
|
||
"type": "toolkit",
|
||
"url": "https://github.com/kharchenkolab/numbat",
|
||
"description": "Haplotype-aware copy number variation inference from single-cell RNA-seq using hidden Markov models.",
|
||
"tags": [
|
||
"preprocessing-tools"
|
||
],
|
||
"tasks": [
|
||
"Preprocessing"
|
||
],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "openmm",
|
||
"name": "OpenMM",
|
||
"type": "toolkit",
|
||
"url": "https://openmm.org/",
|
||
"description": "High-performance toolkit for molecular simulation and GPU-accelerated MD.",
|
||
"tags": [
|
||
"preprocessing-tools"
|
||
],
|
||
"tasks": [
|
||
"Preprocessing"
|
||
],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "rdkit",
|
||
"name": "RDKit",
|
||
"type": "toolkit",
|
||
"url": "https://github.com/rdkit/rdkit",
|
||
"description": "Cheminformatics software & machine learning toolkit.",
|
||
"tags": [
|
||
"preprocessing-tools"
|
||
],
|
||
"tasks": [
|
||
"Preprocessing"
|
||
],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "scanpy",
|
||
"name": "Scanpy",
|
||
"type": "toolkit",
|
||
"url": "https://scanpy.readthedocs.io/en/stable/",
|
||
"description": "Python library for scRNA-seq analysis.",
|
||
"tags": [
|
||
"preprocessing-tools"
|
||
],
|
||
"tasks": [
|
||
"Preprocessing"
|
||
],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "scenic",
|
||
"name": "SCENIC",
|
||
"type": "toolkit",
|
||
"url": "https://github.com/aertslab/SCENIC",
|
||
"description": "Single-cell regulatory network inference and clustering linking transcription factors to co-expressed gene modules.",
|
||
"tags": [
|
||
"preprocessing-tools"
|
||
],
|
||
"tasks": [
|
||
"Preprocessing"
|
||
],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "scipenn",
|
||
"name": "sciPENN",
|
||
"type": "toolkit",
|
||
"url": "https://github.com/jlakkis/sciPENN",
|
||
"description": "RNN-based method for simultaneous protein expression prediction, uncertainty estimation, and cell-type label transfer from CITE-seq and scRNA-seq data.",
|
||
"tags": [
|
||
"preprocessing-tools"
|
||
],
|
||
"tasks": [
|
||
"Preprocessing"
|
||
],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "scvelo",
|
||
"name": "scVelo",
|
||
"type": "toolkit",
|
||
"url": "https://github.com/theislab/scvelo",
|
||
"description": "RNA velocity estimation for single-cell transcriptomics, inferring the direction and speed of cell differentiation.",
|
||
"tags": [
|
||
"preprocessing-tools"
|
||
],
|
||
"tasks": [
|
||
"Preprocessing"
|
||
],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "scvi_tools",
|
||
"name": "scvi-tools",
|
||
"type": "toolkit",
|
||
"url": "https://scvi-tools.org/",
|
||
"description": "Probabilistic models for single-cell omics data analysis.",
|
||
"tags": [
|
||
"preprocessing-tools"
|
||
],
|
||
"tasks": [
|
||
"Preprocessing"
|
||
],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "seqbench",
|
||
"name": "SeqBench",
|
||
"type": "toolkit",
|
||
"url": "https://seqbench.com/",
|
||
"description": "Web-based molecular biology sequence workbench for primer design, cloning simulation (Gibson, Golden Gate, restriction digest), CRISPR guide RNA design, and sequence analysis, with a public REST API, OpenAPI 3.1 spec, and MCP server.",
|
||
"tags": [
|
||
"preprocessing-tools"
|
||
],
|
||
"tasks": [
|
||
"Preprocessing"
|
||
],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "seurat",
|
||
"name": "Seurat",
|
||
"type": "toolkit",
|
||
"url": "https://satijalab.org/seurat/",
|
||
"description": "R library for scRNA-seq analysis.",
|
||
"tags": [
|
||
"preprocessing-tools"
|
||
],
|
||
"tasks": [
|
||
"Preprocessing"
|
||
],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "squidpy",
|
||
"name": "Squidpy",
|
||
"type": "toolkit",
|
||
"url": "https://squidpy.readthedocs.io/",
|
||
"description": "Python library for spatial single-cell analysis.",
|
||
"tags": [
|
||
"preprocessing-tools"
|
||
],
|
||
"tasks": [
|
||
"Preprocessing"
|
||
],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "stagate",
|
||
"name": "STAGATE",
|
||
"type": "toolkit",
|
||
"url": "https://github.com/RucDongLab/STAGATE",
|
||
"description": "Adaptive graph attention auto-encoder for spatial domain identification in spatial transcriptomics.",
|
||
"tags": [
|
||
"preprocessing-tools"
|
||
],
|
||
"tasks": [
|
||
"Preprocessing"
|
||
],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "star",
|
||
"name": "STAR",
|
||
"type": "toolkit",
|
||
"url": "https://github.com/alexdobin/STAR",
|
||
"description": "Ultrafast universal RNA-seq aligner with support for spliced alignment and single-cell quantification via STARsolo.",
|
||
"tags": [
|
||
"preprocessing-tools"
|
||
],
|
||
"tasks": [
|
||
"Preprocessing"
|
||
],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
},
|
||
{
|
||
"id": "tigon",
|
||
"name": "TIGON",
|
||
"type": "toolkit",
|
||
"url": "https://github.com/yutongo/TIGON",
|
||
"description": "Neural optimal transport method for reconstructing growth and dynamic trajectories from single-cell transcriptomics.",
|
||
"tags": [
|
||
"preprocessing-tools"
|
||
],
|
||
"tasks": [
|
||
"Preprocessing"
|
||
],
|
||
"modalities": [],
|
||
"organism": [],
|
||
"api": false
|
||
}
|
||
]
|