--- title: "Awesome Computational Biology - machine-readable resource list" task: "" lineage_type: import upstream_source: https://github.com/inoue0426/awesome-computational-biology/blob/c6f07d90/data/resources.yml upstream_sha: c6f07d90 imported_at: 2026-08-31 prompt_class: catalogue upstream_changes: accepted author: upstream validated: false --- # Awesome Computational Biology - machine-readable resource list # Fields # id : unique slug (required) # name : display name (required) # type : category, e.g. database | tool | model | benchmark | api (required) # url : canonical URL (required) # description : one-line description (required) # license : SPDX identifier or free-text (optional) # api : true | false - whether a programmatic API is available (default: false) # updated : last-known update date as string YYYY-MM-DD (optional) # tasks : list of ML/bio tasks (optional) # modalities : list of data modalities (optional) # tags : additional free-form tags (optional) # organism : list of organisms covered (optional) # paper : DOI or URL to primary publication (optional) resources: - id: chembl_web_services name: "ChEMBL Web Services" type: api url: https://www.ebi.ac.uk/chembl/ws description: "REST API for bioactive molecules, targets, and bioassays." tags: [api] tasks: [] modalities: [] organism: [] api: true - id: clinicaltrials_gov_api name: "ClinicalTrials.gov API" type: api url: https://clinicaltrials.gov/api/gui description: "API for querying clinical trial metadata and results." tags: [api] tasks: [] modalities: [] organism: [] api: true - id: ensembl_rest_api name: "Ensembl REST API" type: api url: https://rest.ensembl.org/ description: "API for genomic annotations, variants, genes, and comparative genomics." tags: [api] tasks: [] modalities: [] organism: [] api: true - id: kegg_rest_api name: "KEGG REST API" type: api url: https://www.kegg.jp/kegg/rest/keggapi.html description: "API for accessing KEGG pathways, compounds, genes, and reactions." tags: [api] tasks: [] modalities: [] organism: [] api: true - id: ncbi_e_utilities name: "NCBI E-utilities" type: api url: https://www.ncbi.nlm.nih.gov/books/NBK25501/ description: "Unified APIs for accessing NCBI databases (Gene, GEO, SRA, PubChem, etc)." tags: [api] tasks: [] modalities: [] organism: [] api: true - id: open_targets_platform_api name: "Open Targets Platform API" type: api url: https://platform.opentargets.org/api description: "API for target–disease associations integrating genetics, genomics, and drug data." tags: [api] tasks: [] modalities: [] organism: [] api: true - id: pubmed_e_utilities_esearch_efetch name: "PubMed E-utilities (esearch/efetch)" type: api url: https://www.nlm.nih.gov/dataguide/edirect/esearch.html description: "APIs for searching and retrieving biomedical literature from PubMed." tags: [api] tasks: [] modalities: [] organism: [] api: true - id: uniprot_rest_api name: "UniProt REST API" type: api url: https://www.uniprot.org/help/api description: "Programmatic access to protein sequence and functional annotation data." tags: [api] tasks: [] modalities: [] organism: [] api: true - id: 1000_genomes_project name: "1000 Genomes Project" type: benchmark url: https://www.internationalgenome.org/ description: "Reference panel of human genetic variation from 2,504 individuals across 26 populations." tags: [benchmarks-and-datasets] tasks: [] modalities: [] organism: [] api: false - id: bace name: "BACE" type: benchmark url: https://www.kaggle.com/datasets/gokturkkoch/bace description: "Binary classification and regression dataset for β-secretase 1 (BACE-1) inhibitor binding affinity." tags: [benchmarks-and-datasets] tasks: [] modalities: [] organism: [] api: false - id: beat_aml name: "BEAT AML" type: benchmark url: https://biodev.github.io/BeatAML2/ description: "Functional ex vivo drug sensitivity measurements paired with genomics for acute myeloid leukemia." tags: [benchmarks-and-datasets] tasks: [] modalities: [] organism: [] api: false - id: bento name: "Bento" type: benchmark url: https://github.com/LigandPro/Bento description: "Protein-ligand docking benchmark covering rigid, flexible, de novo, blind, induced-fit, and covalent docking tasks." tags: [benchmarks-and-datasets] tasks: [] modalities: [] organism: [] api: false - id: bindingdb_curated_sets name: "BindingDB Curated Sets" type: benchmark url: https://www.bindingdb.org/rwd/bind/chemsearch/marvin/SDFdownload.jsp?all_download=yes description: "Curated binding affinity datasets for protein–ligand interaction benchmarking." tags: [benchmarks-and-datasets] tasks: [] modalities: [] organism: [] api: false - id: cancer_therapeutics_response_portal_ctrp name: "Cancer Therapeutics Response Portal (CTRP)" type: benchmark url: https://portals.broadinstitute.org/ctrp/ description: "Drug sensitivity profiles across ~900 cancer cell lines for >400 compounds." tags: [benchmarks-and-datasets] tasks: [] modalities: [] organism: [] api: false - id: clintox name: "ClinTox" type: benchmark url: https://tdcommons.ai/single_pred_tasks/tox/#clintox description: "Clinical toxicity dataset contrasting FDA-approved drugs with those that failed clinical trials due to toxicity." tags: [benchmarks-and-datasets] tasks: [] modalities: [] organism: [] api: false - id: cptac_clinical_proteomic_tumor_analysis_consortium name: "CPTAC (Clinical Proteomic Tumor Analysis Consortium)" type: benchmark url: https://proteomics.cancer.gov/programs/cptac description: "Multi-omic proteogenomic datasets for multiple cancer types linking proteomics with genomics." tags: [benchmarks-and-datasets] tasks: [] modalities: [] organism: [] api: false - id: crossdocked2020 name: "CrossDocked2020" type: benchmark url: https://arxiv.org/abs/2001.01037 description: "Large-scale dataset for structure-based virtual screening." tags: [benchmarks-and-datasets] tasks: [] modalities: [] organism: [] api: false - id: dud_e_directory_of_useful_decoys_enhanced name: "DUD-E (Directory of Useful Decoys, Enhanced)" type: benchmark url: http://dude.docking.org/ description: "Structure-based virtual screening benchmark with active ligands and challenging decoy sets across diverse protein targets." tags: [benchmarks-and-datasets] tasks: [] modalities: [] organism: [] api: false - id: flip_fitness_landscape_inference_for_proteins name: "FLIP (Fitness Landscape Inference for Proteins)" type: benchmark url: https://github.com/J-SNACKKB/FLIP description: "Benchmark collection of protein fitness landscape datasets for evaluating protein ML models." tags: [benchmarks-and-datasets] tasks: [] modalities: [] organism: [] api: false - id: guacamol name: "GuacaMol" type: benchmark url: https://github.com/BenevolentAI/guacamol description: "Benchmark suite for generative molecular design models." tags: [benchmarks-and-datasets] tasks: [] modalities: [] organism: [] api: false - id: hest_xenium_virtual_spatial_transcriptomics name: "HEST Xenium virtual spatial transcriptomics" type: benchmark url: https://huggingface.co/datasets/ratschlab/HEST_Xenium_virtual_spatial_transcriptomics description: "DeepSpot-M predicted transcriptome-wide ST for 59 HEST-1k 10x Xenium samples (~13.3M cells) (gated). Paper: [DeepSpot-M](https://www.medrxiv.org/content/10.64898/2026.06.19.26356060v1)." tags: [benchmarks-and-datasets] tasks: [] modalities: [] organism: [] api: false - id: jump_cell_painting_datasets name: "JUMP Cell Painting Datasets" type: benchmark url: https://github.com/jump-cellpainting/datasets description: "Consortium-scale cell imaging perturbation datasets (chemical and genetic) for phenotypic profiling and drug discovery research." tags: [benchmarks-and-datasets] tasks: [] modalities: [] organism: [] api: false - id: lincs_l1000 name: "LINCS L1000" type: benchmark url: https://lincsproject.org/LINCS/tools/workflows/find-the-best-place-to-obtain-the-lincs-l1000-data description: "Gene expression profiles (978 landmark genes) for >20,000 chemical and genetic perturbations across cell lines." tags: [benchmarks-and-datasets] tasks: [] modalities: [] organism: [] api: false - id: moleculenet name: "MoleculeNet" type: benchmark url: http://moleculenet.ai/ description: "Benchmark datasets for molecular machine learning." tags: [benchmarks-and-datasets] tasks: [] modalities: [] organism: [] api: false - id: moses name: "MOSES" type: benchmark url: https://github.com/molecularsets/moses description: "Benchmarking platform for molecular generation models." tags: [benchmarks-and-datasets] tasks: [] modalities: [] organism: [] api: false - id: ogb_open_graph_benchmark name: "OGB (Open Graph Benchmark)" type: benchmark url: https://ogb.stanford.edu/ description: "Large-scale graph ML benchmark suite including biological datasets such as ogbl-ppa (protein-protein associations) and ogbg-molhiv." tags: [benchmarks-and-datasets] tasks: [] modalities: [] organism: [] api: false - id: openbiolink name: "OpenBioLink" type: benchmark url: https://github.com/OpenBioLink/OpenBioLink description: "Benchmark datasets for biological knowledge graph completion." tags: [benchmarks-and-datasets] tasks: [] modalities: [] organism: [] api: false - id: pharmgkb name: "PharmGKB" type: benchmark url: https://www.pharmgkb.org/ description: "Curated pharmacogenomics dataset linking genetic variants to drug response phenotypes across thousands of drugs." tags: [benchmarks-and-datasets] tasks: [] modalities: [] organism: [] api: false - id: pk_db name: "PK-DB" type: benchmark url: https://pk-db.com/ description: "Open database of experimental pharmacokinetics (PK) and ADME data from clinical and preclinical studies." tags: [benchmarks-and-datasets] tasks: [] modalities: [] organism: [] api: false - id: prism name: "PRISM" type: benchmark url: https://depmap.org/portal/prism/ description: "Cancer drug sensitivity profiling of >4,500 drugs across >900 cancer cell lines using pooled-cell-line barcoding." tags: [benchmarks-and-datasets] tasks: [] modalities: [] organism: [] api: false - id: proteingym name: "ProteinGym" type: benchmark url: https://github.com/OATML-Markslab/ProteinGym description: "Large-scale benchmark of deep mutational scanning assays for evaluating protein fitness landscape models." tags: [benchmarks-and-datasets] tasks: [] modalities: [] organism: [] api: false - id: qm9 name: "QM9" type: benchmark url: https://figshare.com/collections/Quantum_chemistry_structures_and_properties_of_134_kilo_molecules/978904 description: "Quantum chemistry properties for 134K stable small organic molecules computed at DFT level." tags: [benchmarks-and-datasets] tasks: [] modalities: [] organism: [] api: false - id: scib_single_cell_integration_benchmarks name: "scIB (Single-cell Integration Benchmarks)" type: benchmark url: https://github.com/theislab/scib description: "Comprehensive benchmarking framework for single-cell data integration methods." tags: [benchmarks-and-datasets] tasks: [] modalities: [] organism: [] api: false - id: scperturb name: "scPerturb" type: benchmark url: https://github.com/sanderlab/scPerturb description: "Curated and continuously updated single-cell perturbation data resource spanning CRISPR and drug perturbation studies." tags: [benchmarks-and-datasets] tasks: [] modalities: [] organism: [] api: false - id: sider_side_effect_resource name: "SIDER (Side Effect Resource)" type: benchmark url: http://sideeffects.embl.de/ description: "Database of 1,430 approved drugs with their recorded adverse drug reactions across 27 system-organ classes." tags: [benchmarks-and-datasets] tasks: [] modalities: [] organism: [] api: false - id: tabula_muris name: "Tabula Muris" type: benchmark url: https://tabula-muris.ds.czbiohub.org/ description: "Comprehensive single-cell atlas of 20 mouse organs and tissues, enabling cross-tissue and cross-species comparisons." tags: [benchmarks-and-datasets] tasks: [] modalities: [] organism: [] api: false - id: tabula_sapiens name: "Tabula Sapiens" type: benchmark url: https://tabula-sapiens-portal.ds.czbiohub.org/ description: "Comprehensive human single-cell atlas of ~500K cells from 24 organs and tissues across multiple donors." tags: [benchmarks-and-datasets] tasks: [] modalities: [] organism: [] api: false - id: tape_tasks_assessing_protein_embeddings name: "TAPE (Tasks Assessing Protein Embeddings)" type: benchmark url: https://github.com/songlab-cal/tape description: "Benchmark suite of five biologically meaningful semi-supervised learning tasks for evaluating protein representations." tags: [benchmarks-and-datasets] tasks: [] modalities: [] organism: [] api: false - id: tcga_virtual_spatial_transcriptomics_atlas name: "TCGA virtual spatial transcriptomics atlas" type: benchmark url: https://huggingface.co/datasets/ratschlab/TCGA_virtual_spatial_transcriptomics_atlas description: "DeepSpot-M predicted transcriptome-wide ST for TCGA H&E (FF + FFPE; 28,664 slides / 32 cancer types; gated). Paper: [DeepSpot-M](https://www.medrxiv.org/content/10.64898/2026.06.19.26356060v1)." tags: [benchmarks-and-datasets] tasks: [] modalities: [] organism: [] api: false - id: the_cancer_genome_atlas_tcga name: "The Cancer Genome Atlas (TCGA)" type: benchmark url: https://www.cancer.gov/about-nci/organization/ccg/research/structural-genomics/tcga description: "Comprehensive multi-omics (genomics, transcriptomics, proteomics, methylation) dataset for 33 cancer types across ~11,000 patients." tags: [benchmarks-and-datasets] tasks: [] modalities: [] organism: [] api: false - id: therapeutics_data_commons_tdc name: "Therapeutics Data Commons (TDC)" type: benchmark url: https://tdcommons.ai/ description: "Unified benchmark suite covering ADMET, drug-target interaction, drug response, and more." tags: [benchmarks-and-datasets] tasks: [] modalities: [] organism: [] api: false - id: tox21 name: "Tox21" type: benchmark url: https://tripod.nih.gov/tox21/challenge/ description: "12,707 compounds tested in 12 nuclear receptor and stress-response pathway biochemical assays for toxicity prediction." tags: [benchmarks-and-datasets] tasks: [] modalities: [] organism: [] api: false - id: uk_biobank name: "UK Biobank" type: benchmark url: https://www.ukbiobank.ac.uk/ description: "Large-scale biomedical database of ~500K participants with genetic, imaging, and health data for population genetics and disease studies." tags: [benchmarks-and-datasets] tasks: [] modalities: [] organism: [] api: false - id: 10x_genomics_dataset name: "10x Genomics Dataset" type: database url: https://www.10xgenomics.com/resources/datasets description: "Collection of single-cell datasets." tags: [genome] tasks: [] modalities: [Genomics] organism: [] api: false - id: alphafold_protein_structure_database name: "AlphaFold Protein Structure Database" type: database url: https://alphafold.ebi.ac.uk/api-docs description: "3D protein structure predictions." tags: [protein] tasks: [] modalities: [Protein] organism: [] api: false - id: bindingdb name: "BindingDB" type: database url: https://www.bindingdb.org/rwd/bind/index.jsp description: "Compounds and target database." tags: [chemical-protein-interaction, interaction] tasks: [] modalities: [Protein, Small Molecule] organism: [] api: false - id: biocyc name: "BioCyc" type: database url: https://biocyc.org/ description: "Collection of pathway/genome databases across thousands of organisms." tags: [pathway] tasks: [] modalities: [Pathway] organism: [] api: false - id: biogrid name: "BioGRID" type: database url: https://thebiogrid.org/ description: "Protein, genetic, and chemical interactions." tags: [interaction, protein-protein-interaction] tasks: [] modalities: [Protein] organism: [] api: false - id: cancer_cell_line_encyclopedia name: "Cancer Cell Line Encyclopedia" type: database url: https://sites.broadinstitute.org/ccle/ description: "Database of ~1000 cancer cell lines." tags: [drug-cell-line-response, interaction] tasks: [] modalities: [Gene Expression, Small Molecule] organism: [] api: false - id: catalogue_of_somatic_mutations_in_cancer_cosmic name: "Catalogue Of Somatic Mutations In Cancer (COSMIC)" type: database url: https://cancer.sanger.ac.uk/cosmic description: "Resource on somatic mutations in cancers." tags: [genome] tasks: [] modalities: [Genomics] organism: [] api: false - id: cath_database name: "CATH database" type: database url: https://www.cathdb.info/ description: "Hierarchical classification of protein domain structures." tags: [protein] tasks: [] modalities: [Protein] organism: [] api: false - id: cbioportal name: "cBioPortal" type: database url: https://www.cbioportal.org/ description: "Cancer genomics database; aggregating many patient datasets." tags: [genome] tasks: [] modalities: [Genomics] organism: [] api: false - id: cellminer_cross_database_cellminercdb name: "CellMiner Cross Database (CellMinerCDB)" type: database url: https://discover.nci.nih.gov/cellminercdb/ description: "Integrates multiple cancer cell line databases." tags: [drug-cell-line-response, interaction] tasks: [] modalities: [Gene Expression, Small Molecule] organism: [] api: false - id: chebi name: "ChEBI" type: database url: https://www.ebi.ac.uk/chebi/ description: "Database focused on small chemical compounds." tags: [compound] tasks: [] modalities: [Small Molecule] organism: [] api: false - id: chembl name: "ChEMBL" type: database url: https://www.ebi.ac.uk/chembl/ description: "Bioactive molecules with drug-like properties." tags: [compound] tasks: [] modalities: [Small Molecule] organism: [] api: false - id: chemspider name: "ChemSpider" type: database url: http://www.chemspider.com/ description: "Chemical structure database." tags: [compound] tasks: [] modalities: [Small Molecule] organism: [] api: false - id: clinicaltrials_gov name: "ClinicalTrials.gov" type: database url: https://clinicaltrials.gov/ description: "Privately and publicly funded clinical studies." tags: [clinical-trial] tasks: [] modalities: [Clinical] organism: [] api: false - id: comparative_toxicogenomics_database name: "Comparative Toxicogenomics Database" type: database url: http://ctdbase.org/ description: "Chemical-gene interactions, chemical-disease and gene-disease associations, chemical-phenotype associations." tags: [drug-gene-interaction, interaction] tasks: [] modalities: [Gene, Small Molecule] organism: [] api: false - id: critical_assessment_of_structure_prediction_casp name: "Critical Assessment of Structure Prediction (CASP)" type: database url: https://predictioncenter.org/ description: "Assessing methods for protein structure prediction." tags: [protein] tasks: [] modalities: [Protein] organism: [] api: false - id: cz_cellxgene name: "CZ CELLxGENE" type: database url: https://cellxgene.cziscience.com/ description: "Single-cell dataset repository and interactive explorer from the Chan Zuckerberg Initiative." tags: [scrna] tasks: [] modalities: [Single Cell] organism: [] api: false - id: davis_kinase_inhibitors_db name: "Davis kinase inhibitors DB" type: database url: http://staff.cs.utu.fi/~aijrinas/dti/ description: "Experimental kinase inhibitor binding affinity dataset for protein–ligand interaction research." tags: [chemical-protein-interaction, interaction] tasks: [] modalities: [Protein, Small Molecule] organism: [] api: false - id: dependency_map_depmap name: "Dependency Map (DepMap)" type: database url: https://depmap.org/portal/ description: "CRISPR-Cas9 screens in cancer cell lines." tags: [genome] tasks: [] modalities: [Genomics] organism: [] api: false - id: dgidb name: "DGIdb" type: database url: https://www.dgidb.org/ description: "Drug-gene interactions and the druggable genome." tags: [drug-gene-interaction, interaction] tasks: [] modalities: [Gene, Small Molecule] organism: [] api: false - id: diseases name: "DISEASES" type: database url: https://diseases.jensenlab.org/ description: "Gene–disease association database integrating evidence from text mining, curated databases, and experimental data." tags: [disease] tasks: [] modalities: [Disease] organism: [] api: false - id: disgenet name: "DisGeNET" type: database url: https://www.disgenet.org/ description: "Database of gene-disease associations integrating expert-curated and GWAS data." tags: [disease] tasks: [] modalities: [Disease] organism: [] api: false - id: drkg name: "DRKG" type: database url: https://github.com/gnn4dr/DRKG description: "Large-scale biological knowledge graph for drug discovery." tags: [interaction, knowledge-graph] tasks: [] modalities: [Knowledge Graph] organism: [] api: false - id: drug_mechanism_database_drugmechdb name: "Drug Mechanism Database (DrugMechDB)" type: database url: https://github.com/SuLab/DrugMechDB/tree/2.0.1 description: "Mechanisms of action from drug to disease." tags: [interaction, knowledge-graph] tasks: [] modalities: [Knowledge Graph] organism: [] api: false - id: drug_repurposing_hub name: "Drug Repurposing Hub" type: database url: https://repo-hub.broadinstitute.org/repurposing#download-data description: "Collections of drug repurposing data (drug, MoA, target, etc)." tags: [compound] tasks: [] modalities: [Small Molecule] organism: [] api: false - id: drugbank name: "DrugBank" type: database url: https://go.drugbank.com/ description: "Database of drugs and targets (University of Alberta)." tags: [disease] tasks: [] modalities: [Disease] organism: [] api: false - id: drugcentral name: "DrugCentral" type: database url: http://drugcentral.org/ description: "Online drug compendium with drug mode of action and indication information." tags: [compound] tasks: [] modalities: [Small Molecule] organism: [] api: false - id: drugtargetcommons name: "DrugTargetCommons" type: database url: https://drugtargetcommons.fimm.fi/ description: "Community platform for curating and integrating experimental bioactivity data across drugs and targets." tags: [compound] tasks: [] modalities: [Small Molecule] organism: [] api: false - id: encode name: "ENCODE" type: database url: https://www.encodeproject.org/ description: "Encyclopedia of DNA Elements; regulatory and functional genomic elements across the genome." tags: [genome] tasks: [] modalities: [Genomics] organism: [] api: false - id: ensembl name: "Ensembl" type: database url: https://www.ensembl.org/ description: "Genome browser and annotation database for vertebrate and other eukaryotic genomes." tags: [genome] tasks: [] modalities: [Genomics] organism: [] api: false - id: eu_drug_regulating_authorities_clinical_trials_db_eudract name: "EU Drug Regulating Authorities Clinical Trials DB (EudraCT)" type: database url: https://eudract.ema.europa.eu/ description: "European clinical trial database." tags: [clinical-trial] tasks: [] modalities: [Clinical] organism: [] api: false - id: fantom5 name: "FANTOM5" type: database url: https://fantom.gsc.riken.jp/5/ description: "Functional annotation of mammalian genome; comprehensive atlas of active enhancers, promoters, and transcription start sites across human and mouse cell types." tags: [genome] tasks: [] modalities: [Genomics] organism: [] api: false - id: genbank name: "GenBank" type: database url: https://www.ncbi.nlm.nih.gov/genbank/ description: "NCBI's database of genetic sequences." tags: [genome] tasks: [] modalities: [Genomics] organism: [] api: false - id: gene_expression_omnibus name: "Gene Expression Omnibus" type: database url: https://www.ncbi.nlm.nih.gov/geo/ description: "Public functional genomics database." tags: [scrna] tasks: [] modalities: [Single Cell] organism: [] api: false - id: genomics_of_drug_sensitivity_in_cancer_gdsc name: "Genomics of Drug Sensitivity in Cancer (GDSC)" type: database url: https://www.cancerrxgene.org/ description: "Drug sensitivity for ~1000 human cancer cell lines and hundreds of compounds." tags: [benchmarks-and-datasets, drug-cell-line-response, interaction] tasks: [] modalities: [Gene Expression, Small Molecule] organism: [] api: false - id: gnomad name: "gnomAD" type: database url: https://gnomad.broadinstitute.org/ description: "Genome Aggregation Database; genetic variation from large-scale sequencing projects." tags: [genome] tasks: [] modalities: [Genomics] organism: [] api: false - id: hetionet name: "Hetionet" type: database url: https://github.com/hetio/hetionet description: "Heterogeneous network integrating genes, diseases, drugs, pathways, and more." tags: [interaction, knowledge-graph] tasks: [] modalities: [Knowledge Graph] organism: [] api: false - id: hippie name: "HIPPIE" type: database url: http://cbdm-01.zdv.uni-mainz.de/~mschaefer/hippie/ description: "Human protein-protein interaction database." tags: [interaction, protein-protein-interaction] tasks: [] modalities: [Protein] organism: [] api: false - id: hmdb_human_metabolome_database name: "HMDB (Human Metabolome Database)" type: database url: https://hmdb.ca/ description: "Comprehensive database of small molecule metabolites found in the human body." tags: [compound] tasks: [] modalities: [Small Molecule] organism: [] api: false - id: human_cell_atlas name: "Human Cell Atlas" type: database url: https://www.humancellatlas.org/ description: "Open global atlas of all cells in the human body." tags: [scrna] tasks: [] modalities: [Single Cell] organism: [] api: false - id: human_genome_resources_at_ncbi name: "Human Genome Resources at NCBI" type: database url: https://www.ncbi.nlm.nih.gov/projects/genome/guide/human/index.shtml description: "Database for genomics, proteomics, transcriptomics, and systems biology." tags: [genome] tasks: [] modalities: [Genomics] organism: [] api: false - id: human_phenotype_ontology_hpo name: "Human Phenotype Ontology (HPO)" type: database url: https://hpo.jax.org/ description: "Standardized vocabulary of phenotypic abnormalities in human disease, linking genes, variants, and clinical features." tags: [disease] tasks: [] modalities: [Disease] organism: [] api: false - id: icd10 name: "ICD10" type: database url: https://icd.who.int/browse10/2019/en description: "International Classification of Diseases, 10th revision." tags: [clinical-trial] tasks: [] modalities: [Clinical] organism: [] api: false - id: intact name: "IntAct" type: database url: https://www.ebi.ac.uk/intact/home description: "Open-source molecular interaction database and analysis system from EMBL-EBI." tags: [interaction, protein-protein-interaction] tasks: [] modalities: [Protein] organism: [] api: false - id: interpro name: "InterPro" type: database url: https://www.ebi.ac.uk/interpro/ description: "Protein families, domains, and functional sites database integrating 14 member databases including Pfam and PROSITE." tags: [protein] tasks: [] modalities: [Protein] organism: [] api: false - id: jaspar name: "JASPAR" type: database url: http://jaspar.genereg.net/ description: "Database of transcription factor binding profiles." tags: [genome] tasks: [] modalities: [Genomics] organism: [] api: false - id: kegg_compound name: "KEGG COMPOUND" type: database url: https://www.genome.jp/kegg/compound/ description: "Collection of small molecules and biopolymers." tags: [compound] tasks: [] modalities: [Small Molecule] organism: [] api: false - id: kegg_drug name: "KEGG DRUG" type: database url: https://www.genome.jp/kegg/drug/ description: "Comprehensive, approved drug information." tags: [disease] tasks: [] modalities: [Disease] organism: [] api: false - id: kegg_pathway name: "KEGG PATHWAY" type: database url: https://www.genome.jp/kegg/pathway.html description: "Collection of pathway maps." tags: [pathway] tasks: [] modalities: [Pathway] organism: [] api: false - id: kinase_inhibitor_bioactivity_data_kiba name: "Kinase Inhibitor Bioactivity Data (KIBA)" type: database url: https://janeliascicomp.github.io/KIBA/ description: "Integrated bioactivity scores for kinase inhibitors combining Ki, Kd, and IC50 measurements." tags: [chemical-protein-interaction, interaction] tasks: [] modalities: [Protein, Small Molecule] organism: [] api: false - id: lipid_maps name: "LIPID MAPS" type: database url: https://www.lipidmaps.org/databases/lmsd/overview description: "Database of lipids." tags: [compound] tasks: [] modalities: [Small Molecule] organism: [] api: false - id: massbank name: "MassBank" type: database url: http://www.massbank.jp/ description: "Open source databases and tools for mass spectrometry reference spectra." tags: [mass-spectra] tasks: [] modalities: [Mass Spectra] organism: [] api: false - id: mgnify name: "MGnify" type: database url: https://www.ebi.ac.uk/metagenomics/ description: "Resource for metagenomic and metatranscriptomic data." tags: [genome] tasks: [] modalities: [Genomics] organism: [] api: false - id: mimic_iv name: "MIMIC-IV" type: database url: https://mimic.mit.edu/ description: "Freely accessible critical care database." tags: [clinical-trial] tasks: [] modalities: [Clinical] organism: [] api: false - id: mirbase name: "miRBase" type: database url: https://www.mirbase.org/ description: "Reference repository for microRNA gene annotations, sequences, and experimentally validated targets." tags: [gene-regulatory-network, interaction] tasks: [] modalities: [Gene Expression] organism: [] api: false - id: mona_massbank_of_north_america name: "MoNA MassBank of North America" type: database url: https://mona.fiehnlab.ucdavis.edu/ description: "Meta-database of metabolite mass spectra, metadata, and associated compounds." tags: [mass-spectra] tasks: [] modalities: [Mass Spectra] organism: [] api: false - id: msigdb_molecular_signatures_database name: "MSigDB (Molecular Signatures Database)" type: database url: https://www.gsea-msigdb.org/gsea/msigdb description: "Curated gene sets derived from pathways and biological processes." tags: [pathway] tasks: [] modalities: [Pathway] organism: [] api: false - id: nci60 name: "NCI60" type: database url: https://dtp.cancer.gov/discovery_development/nci-60/ description: "Focuses on 60 cancer cell lines and many drugs." tags: [benchmarks-and-datasets, drug-cell-line-response, interaction] tasks: [] modalities: [Gene Expression, Small Molecule] organism: [] api: false - id: nextprot name: "NeXtProt" type: database url: https://www.nextprot.org/ description: "Expert knowledge base on human proteins with deep functional annotation, complementary to UniProt." tags: [protein] tasks: [] modalities: [Protein] organism: [] api: false - id: oadb_observed_antibody_space_database name: "OADB (Observed Antibody Space Database)" type: database url: http://opig.stats.ox.ac.uk/webapps/oas/ description: "Database of antibody sequences from immune repertoire sequencing." tags: [protein] tasks: [] modalities: [Protein] organism: [] api: false - id: omim_online_mendelian_inheritance_in_man name: "OMIM (Online Mendelian Inheritance in Man)" type: database url: https://www.omim.org/ description: "Comprehensive database of human genes and genetic disorders." tags: [disease] tasks: [] modalities: [Disease] organism: [] api: false - id: omnipath name: "OmniPath" type: database url: https://omnipathdb.org/ description: "Comprehensive resource integrating protein interactions, signaling pathways, gene regulatory networks, and miRNA targets from over 100 databases." tags: [pathway] tasks: [] modalities: [Pathway] organism: [] api: false - id: oncokb name: "OncoKB" type: database url: https://www.oncokb.org/ description: "Precision oncology knowledge base of cancer genes, variants, and therapeutic implications." tags: [genome] tasks: [] modalities: [Genomics] organism: [] api: false - id: open_targets_platform name: "Open Targets Platform" type: database url: https://platform.opentargets.org/ description: "Systematic target identification and prioritization platform integrating genetics, genomics, and drug data for drug discovery." tags: [disease] tasks: [] modalities: [Disease] organism: [] api: false - id: pathwaycommons name: "PathwayCommons" type: database url: https://www.pathwaycommons.org/ description: "Database of pathways and interactions." tags: [pathway] tasks: [] modalities: [Pathway] organism: [] api: false - id: pdbbind name: "PDBBind" type: database url: https://www.pdbbind-plus.org.cn/ description: "Binding affinity data for biomolecular complexes." tags: [chemical-protein-interaction, interaction] tasks: [] modalities: [Protein, Small Molecule] organism: [] api: false - id: pfam name: "Pfam" type: database url: https://www.ebi.ac.uk/interpro/entry/pfam/ description: "Database of protein families described by multiple sequence alignments and hidden Markov models." tags: [protein] tasks: [] modalities: [Protein] organism: [] api: false - id: primekg name: "PrimeKG" type: database url: https://github.com/mims-harvard/PrimeKG description: "Multi-modal precision medicine knowledge graph integrating clinical, genetic, and drug data." tags: [interaction, knowledge-graph] tasks: [] modalities: [Knowledge Graph] organism: [] api: false - id: protein_data_bank_pdb name: "PROTEIN DATA BANK (PDB)" type: database url: https://www.rcsb.org/ description: "3D structures of proteins, nucleic acids, complexes." tags: [protein] tasks: [] modalities: [Protein] organism: [] api: false - id: pubchem name: "PubChem" type: database url: https://pubchem.ncbi.nlm.nih.gov/ description: "One of the largest chemical databases (compounds, genes, and proteins)." tags: [compound] tasks: [] modalities: [Small Molecule] organism: [] api: false - id: rcsb_protein_data_bank name: "RCSB Protein Data Bank" type: database url: https://www.rcsb.org/ description: "Repository for structural data of biological molecules." tags: [protein] tasks: [] modalities: [Protein] organism: [] api: false - id: reactome name: "Reactome" type: database url: https://reactome.org/ description: "Expert-curated, peer-reviewed pathway database with detailed reaction mechanisms." tags: [pathway] tasks: [] modalities: [Pathway] organism: [] api: false - id: regnetwork name: "RegNetwork" type: database url: http://www.regnetworkweb.org/ description: "Database of gene regulatory networks covering transcription factor–target gene and miRNA–gene interaction data across multiple species." tags: [gene-regulatory-network, interaction] tasks: [] modalities: [Gene Expression] organism: [] api: false - id: rfam name: "Rfam" type: database url: https://rfam.org/ description: "Database of RNA families with sequence alignments and consensus structures." tags: [genome] tasks: [] modalities: [Genomics] organism: [] api: false - id: rhea name: "Rhea" type: database url: https://www.rhea-db.org/ description: "Database of chemical reactions." tags: [compound] tasks: [] modalities: [Small Molecule] organism: [] api: false - id: roadmap_epigenomics name: "ROADMAP Epigenomics" type: database url: http://www.roadmapepigenomics.org/ description: "Reference epigenome maps for 111 primary human cell types and tissues, including histone modifications, chromatin accessibility, and DNA methylation." tags: [genome] tasks: [] modalities: [Genomics] organism: [] api: false - id: sabdab name: "SAbDab" type: database url: https://opig.stats.ox.ac.uk/webapps/sabdab-sabpred/sabdab description: "Structural Antibody Database containing all antibody structures in the PDB." tags: [protein] tasks: [] modalities: [Protein] organism: [] api: false - id: signor_2_0 name: "SIGNOR 2.0" type: database url: https://signor.uniroma2.it/ description: "Database of causal signaling interactions and pathways, with signed and directed relationships between proteins." tags: [pathway] tasks: [] modalities: [Pathway] organism: [] api: false - id: single_cell_expression_atlas name: "Single Cell Expression Atlas" type: database url: https://www.ebi.ac.uk/gxa/sc/home description: "Public database for single-cell RNA." tags: [scrna] tasks: [] modalities: [Single Cell] organism: [] api: false - id: single_cell_portal name: "Single Cell PORTAL" type: database url: https://singlecell.broadinstitute.org/single_cell description: "Public database for single-cell RNA." tags: [scrna] tasks: [] modalities: [Single Cell] organism: [] api: false - id: snap name: "SNAP" type: database url: https://snap.stanford.edu/biodata/datasets/10002/10002-ChG-Miner.html description: "Dataset of drug-gene interactions." tags: [drug-gene-interaction, interaction] tasks: [] modalities: [Gene, Small Molecule] organism: [] api: false - id: stitch name: "STITCH" type: database url: http://stitch.embl.de/ description: "Chemical-protein interactions." tags: [chemical-protein-interaction, interaction] tasks: [] modalities: [Protein, Small Molecule] organism: [] api: false - id: string name: "STRING" type: database url: https://string-db.org/ description: "PPI networks for multiple organisms." tags: [interaction, protein-protein-interaction] tasks: [] modalities: [Protein] organism: [] api: false - id: the_genotype_tissue_expression_gtex name: "The Genotype-Tissue Expression (GTEx)" type: database url: https://gtexportal.org/home/ description: "Human gene expression and regulation resource." tags: [genome] tasks: [] modalities: [Genomics] organism: [] api: false - id: the_human_protein_atlas name: "THE HUMAN PROTEIN ATLAS" type: database url: https://www.proteinatlas.org/ description: "Comprehensive human protein database (cells, tissues, organs)." tags: [protein] tasks: [] modalities: [Protein] organism: [] api: false - id: therapeutic_target_database name: "Therapeutic Target Database" type: database url: https://idrblab.net/ttd/full-data-download description: "Drug-target, target-disease, and drug-disease datasets." tags: [compound] tasks: [] modalities: [Small Molecule] organism: [] api: false - id: trrust_v2 name: "TRRUST v2" type: database url: https://www.grnpedia.org/trrust/ description: "Manually curated database of human and mouse transcriptional regulatory interactions between transcription factors and their target genes, expanded with literature-derived evidence." tags: [gene-regulatory-network, interaction] tasks: [] modalities: [Gene Expression] organism: [] api: false - id: ucsc_genome_browser name: "UCSC Genome Browser" type: database url: https://genome.ucsc.edu/ description: "UCSC's genome browser." tags: [genome] tasks: [] modalities: [Genomics] organism: [] api: false - id: uniclust name: "Uniclust" type: database url: https://uniclust.mmseqs.com/ description: "Clustered protein sequence databases." tags: [protein] tasks: [] modalities: [Protein] organism: [] api: false - id: uniprot name: "UniProt" type: database url: https://www.uniprot.org/ description: "Functional information on proteins." tags: [protein] tasks: [] modalities: [Protein] organism: [] api: false - id: uniref name: "UniRef" type: database url: https://www.uniprot.org/uniref/ description: "Non-redundant sequence database clustering UniProtKB entries at multiple sequence identity thresholds." tags: [protein] tasks: [] modalities: [Protein] organism: [] api: false - id: wikipathways name: "WikiPathways" type: database url: https://wikipathways.org/ description: "Database of biological pathways." tags: [pathway] tasks: [] modalities: [Pathway] organism: [] api: false - id: zinc_ligand_discovery_database name: "ZINC ligand discovery database" type: database url: https://zinc.docking.org/ description: "Free database of commercially-available compounds for virtual screening." tags: [compound] tasks: [] modalities: [Small Molecule] organism: [] api: false - id: aestetik name: "AESTETIK" type: model url: https://github.com/ratschlab/aestetik description: "Autoencoder for spatial transcriptomics representation learning using topology and histology image knowledge." tags: [foundation-models, single-cell-foundation-models, spatial-foundation-models] tasks: [Foundation Model] modalities: [Single Cell, Spatial Transcriptomics] organism: [] api: false - id: ai4chem_chemllm_7b_chat name: "AI4Chem/ChemLLM-7B-Chat" type: model url: https://huggingface.co/AI4Chem/ChemLLM-7B-Chat description: "LLM for chemical & molecular science." tags: [llm-for-biology] tasks: [Language Modeling] modalities: [Text] organism: [] api: false - id: alphafold3 name: "AlphaFold3" type: model url: https://github.com/google-deepmind/alphafold3 description: "Predicts structures of proteins, nucleic acids, small molecules, and their complexes." tags: [foundation-models, protein-foundation-models, protein-structure-prediction-and-design] tasks: [Foundation Model, Protein Structure Prediction] modalities: [Protein] organism: [] api: false - id: ankh name: "Ankh" type: model url: https://github.com/agemagician/Ankh description: "Efficient protein language model optimized for downstream prediction tasks including secondary structure, localization, and function annotation." tags: [foundation-models, pre-trained-embedding, protein-foundation-models] tasks: [Foundation Model] modalities: [Protein] organism: [] api: false - id: babel name: "BABEL" type: model url: https://github.com/wukevin/babel description: "Cross-modality translation model enabling prediction between scRNA-seq and scATAC-seq profiles without requiring paired single-cell measurements." tags: [foundation-models, multi-omics-foundation-models, single-cell-foundation-models] tasks: [Foundation Model] modalities: [Multi-Omics, Single Cell] organism: [] api: false - id: basenji name: "Basenji" type: model url: https://github.com/calico/basenji description: "Sequential regulatory activity prediction from DNA sequences." tags: [foundation-models, genomics-foundation-models] tasks: [Foundation Model] modalities: [Genomics] organism: [] api: false - id: biogpt name: "BioGPT" type: model url: https://github.com/microsoft/BioGPT description: "LLM for biomedical text generation." tags: [llm-for-biology] tasks: [Language Modeling] modalities: [Text] organism: [] api: false - id: biomedclip name: "BiomedCLIP" type: model url: https://huggingface.co/microsoft/BiomedCLIP-PubMedBERT_256-vit_g_14 description: "CLIP-based vision-language foundation model for biomedical images and text trained on PubMed figure–caption pairs." tags: [foundation-models, multi-modal-foundation-models] tasks: [Foundation Model] modalities: [Multi-Modal] organism: [] api: false - id: biomedlm name: "BioMedLM" type: model url: https://huggingface.co/stanford-crfm/BioMedLM description: "2.7B parameter GPT-2-style language model trained exclusively on biomedical literature from PubMed for biomedical question answering and text generation." tags: [llm-for-biology] tasks: [Language Modeling] modalities: [Text] organism: [] api: false - id: boltz_1 name: "Boltz-1" type: model url: https://github.com/jwohlwend/boltz description: "Open-source all-atom biomolecular structure prediction model for proteins, nucleic acids, small molecules, and their complexes achieving AlphaFold3-level accuracy." tags: [foundation-models, protein-foundation-models, protein-structure-prediction-and-design] tasks: [Foundation Model, Protein Structure Prediction] modalities: [Protein] organism: [] api: false - id: borzoi name: "Borzoi" type: model url: https://github.com/calico/borzoi description: "Extended successor to Enformer for predicting RNA-seq coverage from long genomic sequence windows (524 kb) with improved resolution." tags: [foundation-models, genomics-foundation-models] tasks: [Foundation Model] modalities: [Genomics] organism: [] api: false - id: bulkformer name: "BulkFormer" type: model url: https://github.com/KangBoming/BulkFormer description: "Foundation model for bulk RNA-seq data; learns general transcriptomic representations." tags: [foundation-models, single-cell-foundation-models, transcriptomics-foundation-models] tasks: [Foundation Model] modalities: [Single Cell, Transcriptomics] organism: [] api: false - id: caduceus name: "Caduceus" type: model url: https://github.com/kuleshov-group/caduceus description: "Bidirectional equivariant long-range DNA sequence model based on Mamba." tags: [foundation-models, genomics-foundation-models] tasks: [Foundation Model] modalities: [Genomics] organism: [] api: false - id: cancerfoundation name: "CancerFoundation" type: model url: https://github.com/BoevaLab/CancerFoundation description: "Single-cell RNA-seq foundation model trained exclusively on a curated dataset of malignant cells to learn cancer-specific embeddings." tags: [foundation-models, single-cell-foundation-models, transcriptomics-foundation-models] tasks: [Foundation Model] modalities: [Single Cell, Transcriptomics] organism: [] api: false - id: cassia name: "CASSIA" type: model url: https://github.com/ElliotXie/CASSIA description: "Multi-agent LLM for reference-free, interpretable cell-type annotation of single-cell RNA-seq data, with dedicated annotation, validation, scoring, and reporting agents." tags: [llm-for-biology] tasks: [Language Modeling] modalities: [Text] organism: [] api: false - id: cellot name: "CellOT" type: model url: https://github.com/bunnech/cellot description: "Neural optimal transport framework for predicting single-cell responses to drug and genetic perturbations." tags: [drug-discovery, drug-perturbation] tasks: [Drug Discovery, Drug Perturbation] modalities: [Small Molecule] organism: [] api: false - id: cellplm name: "CellPLM" type: model url: https://github.com/OmicsML/CellPLM description: "Cell pre-trained language model with inter-cell transformer architecture for diverse single-cell analysis tasks." tags: [foundation-models, single-cell-foundation-models, transcriptomics-foundation-models] tasks: [Foundation Model] modalities: [Single Cell, Transcriptomics] organism: [] api: false - id: chai_1 name: "Chai-1" type: model url: https://github.com/chaidiscovery/chai-lab description: "Unified molecular structure prediction model covering proteins, nucleic acids, small molecules, and complexes." tags: [foundation-models, protein-foundation-models, protein-structure-prediction-and-design] tasks: [Foundation Model, Protein Structure Prediction] modalities: [Protein] organism: [] api: false - id: chatdrug name: "ChatDrug" type: model url: https://github.com/chao1224/ChatDrug description: "LLM-based conversational pipeline for drug discovery, using natural language prompts for iterative drug editing and optimization." tags: [llm-for-biology] tasks: [Language Modeling] modalities: [Text] organism: [] api: false - id: chemberta_2 name: "ChemBERTa-2" type: model url: https://github.com/seyonechithrananda/bert-loves-chemistry description: "RoBERTa-based molecular language model pretrained on SMILES for small-molecule representation learning." tags: [compound-embedding, compound-foundation-models, foundation-models] tasks: [Foundation Model] modalities: [Small Molecule] organism: [] api: false - id: chemcpa name: "chemCPA" type: model url: https://github.com/theislab/chemCPA description: "Compositional perturbation autoencoder for predicting single-cell transcriptional responses to unseen drug perturbations and dose combinations." tags: [drug-discovery, drug-perturbation] tasks: [Drug Discovery, Drug Perturbation] modalities: [Small Molecule] organism: [] api: false - id: chief name: "CHIEF" type: model url: https://github.com/hms-dbmi/CHIEF description: "Clinical Histopathology Imaging Evaluation Foundation model integrating histology images and clinical context for pan-cancer analysis." tags: [foundation-models, multi-modal-foundation-models] tasks: [Foundation Model] modalities: [Multi-Modal] organism: [] api: false - id: clawbio name: "ClawBio" type: model url: https://github.com/ClawBio/ClawBio description: "Bioinformatics-native AI agent skill library with local-first pharmacogenomics, ancestry PCA, semantic similarity, nutrigenomics, and metagenomics skills." tags: [llm-for-biology] tasks: [Language Modeling] modalities: [Text] organism: [] api: false - id: cmonge name: "CMonge" type: model url: https://github.com/AI4SCR/conditional-monge-gap description: "Conditional optimal transport model for generalizable single-cell perturbation response prediction across drugs and doses." tags: [drug-discovery, drug-perturbation] tasks: [Drug Discovery, Drug Perturbation] modalities: [Small Molecule] organism: [] api: false - id: concerto name: "Concerto" type: model url: https://github.com/melobio/Concerto-reproducibility description: "Contrastive self-supervised learning framework for single-cell multimodal data integration, batch correction, and reference-query mapping." tags: [foundation-models, multi-omics-foundation-models, single-cell-foundation-models] tasks: [Foundation Model] modalities: [Multi-Omics, Single Cell] organism: [] api: false - id: conch name: "CONCH" type: model url: https://github.com/mahmoodlab/CONCH description: "Vision-language foundation model for computational pathology trained with contrastive captioning on pathology image–text pairs." tags: [foundation-models, single-cell-foundation-models, spatial-foundation-models] tasks: [Foundation Model] modalities: [Single Cell, Spatial Transcriptomics] organism: [] api: false - id: cyclecdr name: "cycleCDR" type: model url: https://github.com/hliulab/cycleCDR description: "Interpretable cycle-consistency framework for modeling cellular responses to drug perturbations." tags: [drug-discovery, drug-perturbation] tasks: [Drug Discovery, Drug Perturbation] modalities: [Small Molecule] organism: [] api: false - id: deepaeg name: "DeepAEG" type: model url: https://github.com/zhejiangzhuque/DeepAEG description: "GNN embedding + attention mechanism." tags: [drug-discovery, drug-response-prediction] tasks: [Drug Discovery, Drug Response Prediction] modalities: [Small Molecule] organism: [] api: false - id: deepdsc name: "DeepDSC" type: model url: https://ieeexplore-ieee-org.ezp2.lib.umn.edu/stamp/stamp.jsp?tp=&arnumber=8723620&tag=1 description: "Autoencoder + fully connected NN." tags: [drug-discovery, drug-response-prediction] tasks: [Drug Discovery, Drug Response Prediction] modalities: [Small Molecule] organism: [] api: false - id: deepdta name: "DeepDTA" type: model url: https://github.com/hkmztrk/DeepDTA description: "Deep learning model using CNNs on protein sequences and drug SMILES." tags: [drug-discovery, drug-target-interaction] tasks: [Drug Discovery, Drug Target Interaction] modalities: [Protein, Small Molecule] organism: [] api: false - id: deeppurpose name: "DeepPurpose" type: model url: https://github.com/kexinhuang12345/DeepPurpose description: "Deep learning library for drug repurposing." tags: [drug-discovery, drug-repurposing] tasks: [Drug Discovery, Drug Repurposing] modalities: [Small Molecule] organism: [] api: false - id: deepsea name: "DeepSEA" type: model url: http://deepsea.princeton.edu/ description: "Deep learning framework for predicting chromatin effects of sequence alterations with single-nucleotide sensitivity across thousands of chromatin features." tags: [foundation-models, genomics-foundation-models] tasks: [Foundation Model] modalities: [Genomics] organism: [] api: false - id: deepspot name: "DeepSpot" type: model url: https://github.com/ratschlab/DeepSpot description: "Deep learning model predicting spatial transcriptomics from H&E images at spot and single-cell resolution." tags: [foundation-models, single-cell-foundation-models, spatial-foundation-models] tasks: [Foundation Model] modalities: [Single Cell, Spatial Transcriptomics] organism: [] api: false - id: deepspot_m name: "DeepSpot-M" type: model url: https://github.com/ratschlab/DeepSpotM description: "Multimodal foundation model for transcriptome-wide virtual spatial transcriptomics from histology." tags: [foundation-models, single-cell-foundation-models, spatial-foundation-models] tasks: [Foundation Model] modalities: [Single Cell, Spatial Transcriptomics] organism: [] api: false - id: deepspot2cell name: "DeepSpot2Cell" type: model url: https://github.com/ratschlab/DeepSpot2Cell description: "Predicts virtual single-cell spatial transcriptomics from H&E using spot-level supervision (NeurIPS 2025 Imageomics)." tags: [foundation-models, single-cell-foundation-models, spatial-foundation-models] tasks: [Foundation Model] modalities: [Single Cell, Spatial Transcriptomics] organism: [] api: false - id: dgdrp name: "DGDRP" type: model url: https://github.com/minwoopak/heteronet description: "Multi-view embedding neural network." tags: [drug-discovery, drug-response-prediction] tasks: [Drug Discovery, Drug Response Prediction] modalities: [Small Molecule] organism: [] api: false - id: diffdock name: "DiffDock" type: model url: https://github.com/gcorso/DiffDock description: "Diffusion generative model for molecular docking, predicting the binding pose of small molecules to protein targets." tags: [drug-discovery, molecular-generation] tasks: [Drug Discovery, Molecular Generation] modalities: [Small Molecule] organism: [] api: false - id: diffsbdd name: "DiffSBDD" type: model url: https://github.com/arneschneuing/DiffSBDD description: "Equivariant diffusion model for structure-based drug design that generates molecules and binding conformations for protein targets." tags: [drug-discovery, molecular-generation] tasks: [Drug Discovery, Molecular Generation] modalities: [Small Molecule] organism: [] api: false - id: dnabert name: "DNABERT" type: model url: https://github.com/jerryji1993/DNABERT description: "Pre-trained bidirectional encoder for DNA sequence analysis." tags: [foundation-models, genomics-foundation-models] tasks: [Foundation Model] modalities: [Genomics] organism: [] api: false - id: dnabert_2 name: "DNABERT-2" type: model url: https://github.com/Zhihan1996/DNABERT_2 description: "Improved genome foundation model with efficient tokenization." tags: [foundation-models, genomics-foundation-models] tasks: [Foundation Model] modalities: [Genomics] organism: [] api: false - id: drgat name: "drGAT" type: model url: https://github.com/inoue0426/drGAT description: "Attention-based model for drug response prediction with gene explainability." tags: [drug-discovery, drug-response-prediction] tasks: [Drug Discovery, Drug Response Prediction] modalities: [Small Molecule] organism: [] api: false - id: drugban name: "DrugBAN" type: model url: https://github.com/peizhenbai/DrugBAN description: "Bilinear attention network for interpretable DTI prediction." tags: [drug-discovery, drug-target-interaction] tasks: [Drug Discovery, Drug Target Interaction] modalities: [Protein, Small Molecule] organism: [] api: false - id: druml name: "DRUML" type: model url: https://github.com/CutillasLab/DRUMLR description: "Ensemble machine learning framework combining standard ML with deep learning to systematically rank anti-cancer drugs from proteomics and RNA-seq data." tags: [drug-discovery, drug-response-prediction] tasks: [Drug Discovery, Drug Response Prediction] modalities: [Small Molecule] organism: [] api: false - id: dtinet name: "DTINet" type: model url: https://github.com/luoyunan/DTINet description: "Network-based framework integrating heterogeneous biological data for DTI prediction." tags: [drug-discovery, drug-target-interaction] tasks: [Drug Discovery, Drug Target Interaction] modalities: [Protein, Small Molecule] organism: [] api: false - id: enformer name: "Enformer" type: model url: https://github.com/deepmind/deepmind-research/tree/master/enformer description: "Transformer model predicting gene expression from DNA sequence." tags: [foundation-models, genomics-foundation-models] tasks: [Foundation Model] modalities: [Genomics] organism: [] api: false - id: esm3 name: "ESM3" type: model url: https://github.com/evolutionaryscale/esm description: "Multimodal protein language model that jointly reasons over sequence, structure, and function for generative protein design and engineering." tags: [foundation-models, protein-foundation-models, protein-structure-prediction-and-design] tasks: [Foundation Model, Protein Structure Prediction] modalities: [Protein] organism: [] api: false - id: esmfold name: "ESMFold" type: model url: https://github.com/facebookresearch/esm description: "Fast protein structure prediction using language model embeddings." tags: [foundation-models, protein-foundation-models, protein-structure-prediction-and-design] tasks: [Foundation Model, Protein Structure Prediction] modalities: [Protein] organism: [] api: false - id: evo name: "Evo" type: model url: https://github.com/evo-design/evo description: "Long-context genomic foundation model (up to 1M tokens)." tags: [foundation-models, genomics-foundation-models] tasks: [Foundation Model] modalities: [Genomics] organism: [] api: false - id: evodiff name: "EvoDiff" type: model url: https://github.com/microsoft/evodiff description: "Discrete diffusion framework for protein sequence generation trained on evolutionary-scale data, supporting unconditional generation, disordered region design, and functional motif scaffolding. [ [paper-2023](https://www.biorxiv.org/content/10.1101/2023.09.11.556673v1) ]" tags: [foundation-models, protein-foundation-models, protein-structure-prediction-and-design] tasks: [Foundation Model, Protein Structure Prediction] modalities: [Protein] organism: [] api: false - id: evolutionary_scale_modeling_esm name: "Evolutionary Scale Modeling (ESM)" type: model url: https://github.com/facebookresearch/esm description: "Protein embeddings." tags: [foundation-models, pre-trained-embedding, protein-foundation-models] tasks: [Foundation Model] modalities: [Protein] organism: [] api: false - id: gears name: "GEARS" type: model url: https://github.com/snap-stanford/GEARS description: "Graph-based model for predicting transcriptional responses to single and combinatorial genetic perturbations using biological priors." tags: [foundation-models, single-cell-foundation-models, transcriptomics-foundation-models] tasks: [Foundation Model] modalities: [Single Cell, Transcriptomics] organism: [] api: false - id: genecompass name: "GeneCompass" type: model url: https://github.com/xCompass-AI/GeneCompass description: "Large-scale foundation model integrating DNA regulatory sequences and single-cell transcriptomics from 120M+ cells across multiple species for gene regulation prediction." tags: [foundation-models, multi-omics-foundation-models, single-cell-foundation-models] tasks: [Foundation Model] modalities: [Multi-Omics, Single Cell] organism: [] api: false - id: geneformer name: "Geneformer" type: model url: https://huggingface.co/ctheodoris/Geneformer description: "Context-aware, attention-based deep learning model pretrained on a large corpus of single-cell transcriptomes." tags: [foundation-models, single-cell-foundation-models, transcriptomics-foundation-models] tasks: [Foundation Model] modalities: [Single Cell, Transcriptomics] organism: [] api: false - id: genegpt name: "GeneGPT" type: model url: https://github.com/ncbi/GeneGPT description: "LLM for biomedical information, integrated with various APIs." tags: [llm-for-biology] tasks: [Language Modeling] modalities: [Text] organism: [] api: false - id: genept name: "GenePT" type: model url: https://github.com/yiqunchen/GenePT description: "Foundation LLM for single-cell data." tags: [llm-for-biology] tasks: [Language Modeling] modalities: [Text] organism: [] api: false - id: gigapath name: "GigaPath" type: model url: https://github.com/prov-gigapath/prov-gigapath description: "Slide-level digital pathology foundation model pretrained on 1.3 billion pathology image tokens from whole-slide images." tags: [foundation-models, single-cell-foundation-models, spatial-foundation-models] tasks: [Foundation Model] modalities: [Single Cell, Spatial Transcriptomics] organism: [] api: false - id: glue name: "GLUE" type: model url: https://github.com/gao-lab/GLUE description: "Graph-Linked Unified Embedding framework for unpaired single-cell multi-omics data integration across RNA, ATAC, methylation, and protein modalities." tags: [foundation-models, multi-omics-foundation-models, single-cell-foundation-models] tasks: [Foundation Model] modalities: [Multi-Omics, Single Cell] organism: [] api: false - id: gpn_genomic_pre_trained_network name: "GPN (Genomic Pre-trained Network)" type: model url: https://github.com/songlab-cal/gpn description: "Masked language model for DNA sequences enabling zero-shot variant effect prediction without requiring functional annotations." tags: [foundation-models, genomics-foundation-models] tasks: [Foundation Model] modalities: [Genomics] organism: [] api: false - id: graphdta name: "GraphDTA" type: model url: https://github.com/thinng/GraphDTA description: "Graph neural network–based DTI prediction using molecular graphs." tags: [drug-discovery, drug-target-interaction] tasks: [Drug Discovery, Drug Target Interaction] modalities: [Protein, Small Molecule] organism: [] api: false - id: grover name: "GROVER" type: model url: https://github.com/tencent-ailab/grover description: "Self-supervised graph transformer for large-scale molecular representation learning from unlabeled compounds." tags: [compound-embedding, compound-foundation-models, foundation-models] tasks: [Foundation Model] modalities: [Small Molecule] organism: [] api: false - id: hidra name: "HiDRA" type: model url: https://github.com/bsml320/HiDRA description: "Hierarchical network model incorporating gene and pathway-level information for cancer drug response prediction." tags: [drug-discovery, drug-response-prediction] tasks: [Drug Discovery, Drug Response Prediction] modalities: [Small Molecule] organism: [] api: false - id: hyenadna name: "HyenaDNA" type: model url: https://github.com/HazyResearch/hyena-dna description: "Long-range genomic foundation model handling sequences up to 1M tokens with sub-quadratic attention." tags: [foundation-models, genomics-foundation-models] tasks: [Foundation Model] modalities: [Genomics] organism: [] api: false - id: jamie name: "JAMIE" type: model url: https://github.com/Oafish1/JAMIE description: "Joint variational autoencoder for multimodal single-cell data imputation and embedding." tags: [foundation-models, multi-omics-foundation-models, single-cell-foundation-models] tasks: [Foundation Model] modalities: [Multi-Omics, Single Cell] organism: [] api: false - id: jtvae name: "JTVAE" type: model url: https://github.com/wengong-jin/icml18-jtnn description: "Junction tree variational autoencoder for molecular graph generation that guarantees chemical validity via a hierarchical tree decomposition." tags: [drug-discovery, molecular-generation] tasks: [Drug Discovery, Molecular Generation] modalities: [Small Molecule] organism: [] api: false - id: matcha name: "Matcha" type: model url: https://github.com/LigandPro/Matcha description: "Multi-stage Riemannian flow matching model for physically valid molecular docking with scoring, pose filtering, and benchmarks." tags: [drug-discovery, molecular-generation] tasks: [Drug Discovery, Molecular Generation] modalities: [Small Molecule] organism: [] api: false - id: mcpinn name: "MCPINN" type: model url: https://github.com/mhlee0903/multi_channels_PINN description: "Drug discovery via compound-protein interaction and machine learning." tags: [compound-protein-interaction, drug-discovery] tasks: [Compound-Protein Interaction, Drug Discovery] modalities: [Protein, Small Molecule] organism: [] api: false - id: midas name: "MIDAS" type: model url: https://github.com/labomics/midas description: "Mosaic integration and differential accessibility model for single-cell multi-omics that handles arbitrary missing-modality combinations across transcriptomics, chromatin accessibility, and proteomics." tags: [foundation-models, multi-omics-foundation-models, single-cell-foundation-models] tasks: [Foundation Model] modalities: [Multi-Omics, Single Cell] organism: [] api: false - id: mira name: "MIRA" type: model url: https://github.com/cistrome/MIRA description: "Probabilistic multimodal topic model jointly modeling single-cell transcriptomics and chromatin accessibility for regulatory network inference." tags: [foundation-models, multi-omics-foundation-models, single-cell-foundation-models] tasks: [Foundation Model] modalities: [Multi-Omics, Single Cell] organism: [] api: false - id: mofa name: "MOFA+" type: model url: https://github.com/bioFAM/MOFA2 description: "Multi-Omics Factor Analysis framework identifying shared axes of variation across bulk and single-cell datasets including RNA, ATAC, proteomics, methylation, and copy number." tags: [foundation-models, multi-omics-foundation-models, single-cell-foundation-models] tasks: [Foundation Model] modalities: [Multi-Omics, Single Cell] organism: [] api: false - id: mofgcn name: "MOFGCN" type: model url: https://github.com/weiba/MOFGCN/tree/main description: "GCN + heterogeneous network." tags: [drug-discovery, drug-response-prediction] tasks: [Drug Discovery, Drug Response Prediction] modalities: [Small Molecule] organism: [] api: false - id: mol2vec name: "Mol2Vec" type: model url: https://github.com/samoturk/mol2vec description: "Unsupervised molecular embedding method inspired by Word2Vec for learning vector representations of chemical substructures." tags: [compound-embedding, compound-foundation-models, foundation-models] tasks: [Foundation Model] modalities: [Small Molecule] organism: [] api: false - id: molecular_transformer name: "Molecular Transformer" type: model url: https://github.com/pschwllr/MolecularTransformer description: "Sequence-to-sequence model for retrosynthesis prediction." tags: [drug-discovery, molecular-generation] tasks: [Drug Discovery, Molecular Generation] modalities: [Small Molecule] organism: [] api: false - id: molformer name: "MolFormer" type: model url: https://github.com/IBM/molformer description: "Linear attention transformer pretrained on millions of SMILES strings for efficient molecular embeddings." tags: [compound-embedding, compound-foundation-models, foundation-models] tasks: [Foundation Model] modalities: [Small Molecule] organism: [] api: false - id: molgpt name: "MolGPT" type: model url: https://github.com/devalab/molgpt description: "Transformer-based model for molecular generation." tags: [drug-discovery, molecular-generation] tasks: [Drug Discovery, Molecular Generation] modalities: [Small Molecule] organism: [] api: false - id: molt5 name: "MolT5" type: model url: https://github.com/blender-nlp/MolT5 description: "Language model for molecular tasks bridging text and SMILES, enabling molecule captioning and text-driven molecule generation." tags: [llm-for-biology] tasks: [Language Modeling] modalities: [Text] organism: [] api: false - id: moltrans name: "MolTrans" type: model url: https://github.com/kexinhuang12345/MolTrans description: "Transformer-based DTI model leveraging molecular substructures." tags: [drug-discovery, drug-target-interaction] tasks: [Drug Discovery, Drug Target Interaction] modalities: [Protein, Small Molecule] organism: [] api: false - id: multigrate name: "Multigrate" type: model url: https://github.com/theislab/multigrate description: "Asymmetric multi-omics variational autoencoder for integrating single-cell data across RNA, ATAC, and protein modalities with missing-modality support." tags: [foundation-models, multi-omics-foundation-models, single-cell-foundation-models] tasks: [Foundation Model] modalities: [Multi-Omics, Single Cell] organism: [] api: false - id: multivi name: "MultiVI" type: model url: https://github.com/scverse/scvi-tools description: "Multi-modal variational autoencoder for integrating paired and unpaired single-cell RNA-seq and ATAC-seq measurements into a unified latent space." tags: [foundation-models, multi-omics-foundation-models, single-cell-foundation-models] tasks: [Foundation Model] modalities: [Multi-Omics, Single Cell] organism: [] api: false - id: musk name: "MUSK" type: model url: https://github.com/lilab-stanford/MUSK description: "Vision-language foundation model for precision oncology analyzing multimodal paired text and pathology image data for biomarker prediction and retrieval." tags: [foundation-models, multi-modal-foundation-models] tasks: [Foundation Model] modalities: [Multi-Modal] organism: [] api: false - id: nbbayeslm name: "NbBayesLM" type: model url: https://github.com/FairuzShadmaniShishir/NbBayesLM description: "Bayesian neural network integrating protein language model embeddings and physicochemical features to predict nanobody thermostability with uncertainty estimates. [Paper](https://www.frontiersin.org/journals/bioinformatics/articles/10.3389/fbinf.2026.1832968/full)" tags: [protein-property-prediction] tasks: [Protein Property Prediction] modalities: [Protein] organism: [] api: false - id: neodti name: "NeoDTI" type: model url: https://github.com/FangpingWan/NeoDTI description: "Library for drug-target interaction prediction." tags: [drug-discovery, drug-target-interaction] tasks: [Drug Discovery, Drug Target Interaction] modalities: [Protein, Small Molecule] organism: [] api: false - id: nicheformer name: "Nicheformer" type: model url: https://github.com/theislab/nicheformer description: "Foundation model for single-cell and spatial omics using a transformer architecture with positional embeddings to encode spatial cell information." tags: [foundation-models, single-cell-foundation-models, spatial-foundation-models] tasks: [Foundation Model] modalities: [Single Cell, Spatial Transcriptomics] organism: [] api: false - id: nucleotide_transformer name: "Nucleotide Transformer" type: model url: https://github.com/instadeepai/nucleotide-transformer description: "Foundation model for genomic sequences across multiple species." tags: [foundation-models, genomics-foundation-models] tasks: [Foundation Model] modalities: [Genomics] organism: [] api: false - id: omegafold name: "OmegaFold" type: model url: https://github.com/HeliXonProtein/OmegaFold description: "High-resolution de novo protein structure prediction from sequence." tags: [foundation-models, protein-foundation-models, protein-structure-prediction-and-design] tasks: [Foundation Model, Protein Structure Prediction] modalities: [Protein] organism: [] api: false - id: openfold name: "OpenFold" type: model url: https://github.com/aqlaboratory/openfold description: "Trainable, memory-efficient open-source reproduction of AlphaFold2 enabling custom protein structure prediction workflows." tags: [foundation-models, protein-foundation-models, protein-structure-prediction-and-design] tasks: [Foundation Model, Protein Structure Prediction] modalities: [Protein] organism: [] api: false - id: paccmannrl name: "PaccMannRL" type: model url: https://github.com/PaccMann/paccmann_generator description: "Reinforcement learning-based generative model for de novo hit-like anticancer molecule design from transcriptomic data." tags: [drug-discovery, molecular-generation] tasks: [Drug Discovery, Molecular Generation] modalities: [Small Molecule] organism: [] api: false - id: pathomicfusion name: "PathomicFusion" type: model url: https://github.com/mahmoodlab/PathomicFusion description: "Integrated framework fusing histopathology and genomic features via CNN, GNN, and attention gating for cancer diagnosis and prognosis." tags: [foundation-models, multi-modal-foundation-models] tasks: [Foundation Model] modalities: [Multi-Modal] organism: [] api: false - id: phikon name: "Phikon" type: model url: https://huggingface.co/owkin/phikon description: "ViT-based pathology foundation model pretrained with iBOT self-supervision on TCGA whole-slide images." tags: [foundation-models, single-cell-foundation-models, spatial-foundation-models] tasks: [Foundation Model] modalities: [Single Cell, Spatial Transcriptomics] organism: [] api: false - id: plip name: "PLIP" type: model url: https://github.com/PathologyFoundation/plip description: "Vision-language foundation model for pathology trained with contrastive learning on pathology image–text pairs for image classification and text-to-image retrieval." tags: [foundation-models, multi-modal-foundation-models] tasks: [Foundation Model] modalities: [Multi-Modal] organism: [] api: false - id: porpoise name: "PORPOISE" type: model url: https://github.com/mahmoodlab/PORPOISE description: "Pan-cancer integrative histology-genomic analysis framework using multimodal deep learning for patient stratification." tags: [foundation-models, multi-modal-foundation-models] tasks: [Foundation Model] modalities: [Multi-Modal] organism: [] api: false - id: prnet name: "PRNet" type: model url: https://github.com/Perturbation-Response-Prediction/PRnet description: "Deep generative model for predicting transcriptional responses to novel chemical perturbations for drug discovery." tags: [drug-discovery, drug-perturbation] tasks: [Drug Discovery, Drug Perturbation] modalities: [Small Molecule] organism: [] api: false - id: progen2 name: "ProGen2" type: model url: https://github.com/salesforce/progen description: "Protein language model trained on diverse protein families for sequence generation and fitness prediction." tags: [foundation-models, pre-trained-embedding, protein-foundation-models] tasks: [Foundation Model] modalities: [Protein] organism: [] api: false - id: proteinmpnn name: "ProteinMPNN" type: model url: https://github.com/dauparas/ProteinMPNN description: "Deep learning model for protein sequence design given backbone structure." tags: [foundation-models, protein-foundation-models, protein-structure-prediction-and-design] tasks: [Foundation Model, Protein Structure Prediction] modalities: [Protein] organism: [] api: false - id: prottrans name: "ProtTrans" type: model url: https://github.com/agemagician/ProtTrans description: "Suite of protein language models (ProtBERT, ProtT5, ProtXLNet) trained on billions of protein sequences from UniRef and BFD." tags: [foundation-models, pre-trained-embedding, protein-foundation-models] tasks: [Foundation Model] modalities: [Protein] organism: [] api: false - id: recover name: "RECOVER" type: model url: https://github.com/RECOVERcoalition/Recover description: "Machine learning framework for predicting synergistic drug combination responses across cell lines." tags: [drug-discovery, drug-response-prediction] tasks: [Drug Discovery, Drug Response Prediction] modalities: [Small Molecule] organism: [] api: false - id: reinvent name: "REINVENT" type: model url: https://github.com/MolecularAI/Reinvent description: "Reinforcement learning for de novo drug design." tags: [drug-discovery, molecular-generation] tasks: [Drug Discovery, Molecular Generation] modalities: [Small Molecule] organism: [] api: false - id: release name: "ReLeaSE" type: model url: https://github.com/isayev/ReLeaSE description: "Deep reinforcement learning framework for de novo drug design combining a generative and predictive model." tags: [drug-discovery, molecular-generation] tasks: [Drug Discovery, Molecular Generation] modalities: [Small Molecule] organism: [] api: false - id: rfdiffusion name: "RFdiffusion" type: model url: https://github.com/RosettaCommons/RFdiffusion description: "Generative model for protein backbone design using diffusion." tags: [foundation-models, protein-foundation-models, protein-structure-prediction-and-design] tasks: [Foundation Model, Protein Structure Prediction] modalities: [Protein] organism: [] api: false - id: rosettafold name: "RoseTTAFold" type: model url: https://github.com/RosettaCommons/RoseTTAFold description: "Three-track neural network for protein structure prediction." tags: [foundation-models, protein-foundation-models, protein-structure-prediction-and-design] tasks: [Foundation Model, Protein Structure Prediction] modalities: [Protein] organism: [] api: false - id: saprot name: "SaProt" type: model url: https://github.com/westlake-reup/SaProt description: "Structure-aware protein language model using structure-aware tokens that encode both sequence and backbone geometry for improved function prediction." tags: [foundation-models, protein-foundation-models, protein-structure-prediction-and-design] tasks: [Foundation Model, Protein Structure Prediction] modalities: [Protein] organism: [] api: false - id: saturn name: "SATURN" type: model url: https://github.com/snap-stanford/SATURN description: "Transformer-based model integrating gene expression and protein sequences via a protein language model to learn unified multi-species cell embeddings." tags: [foundation-models, single-cell-foundation-models, transcriptomics-foundation-models] tasks: [Foundation Model] modalities: [Single Cell, Transcriptomics] organism: [] api: false - id: scarches name: "scArches" type: model url: https://github.com/theislab/scarches description: "Transfer learning framework for mapping new single-cell datasets onto pre-trained reference atlases across batches, conditions, and modalities." tags: [domain-alignment, foundation-models, single-cell-foundation-models] tasks: [Domain Alignment, Foundation Model] modalities: [Single Cell] organism: [] api: false - id: scbert name: "scBERT" type: model url: https://github.com/TencentAILabHealthcare/scBERT description: "BERT-based foundation model pretrained on large-scale scRNA-seq data for cell type annotation." tags: [foundation-models, single-cell-foundation-models, transcriptomics-foundation-models] tasks: [Foundation Model] modalities: [Single Cell, Transcriptomics] organism: [] api: false - id: scbutterfly name: "scButterfly" type: model url: https://github.com/BioX-NKU/scButterfly description: "Dual-aligned variational autoencoder for single-cell cross-modality translation between paired and unpaired multiomics data." tags: [foundation-models, multi-omics-foundation-models, single-cell-foundation-models] tasks: [Foundation Model] modalities: [Multi-Omics, Single Cell] organism: [] api: false - id: scfoundation name: "scFoundation" type: model url: https://github.com/biomap-research/scFoundation description: "Large-scale foundation model for single-cell gene expression, enabling multiple downstream tasks." tags: [foundation-models, single-cell-foundation-models, transcriptomics-foundation-models] tasks: [Foundation Model] modalities: [Single Cell, Transcriptomics] organism: [] api: false - id: scgpt name: "scGPT" type: model url: https://github.com/bowang-lab/scGPT description: "Transformer-based foundation model pretrained on millions of single-cell profiles." tags: [foundation-models, single-cell-foundation-models, transcriptomics-foundation-models] tasks: [Foundation Model] modalities: [Single Cell, Transcriptomics] organism: [] api: false - id: scgpt_spatial name: "scGPT-spatial" type: model url: https://github.com/bowang-lab/scGPT-spatial description: "Extension of scGPT for spatial transcriptomics with continual pretraining and a mixture-of-experts decoder for spatial gene expression analysis." tags: [foundation-models, single-cell-foundation-models, spatial-foundation-models] tasks: [Foundation Model] modalities: [Single Cell, Spatial Transcriptomics] organism: [] api: false - id: scmulan name: "scMulan" type: model url: https://github.com/SuperBianC/scMulan description: "Single-cell multi-omic language model pretrained on ~10M cells spanning transcriptomics, epigenomics, and proteomics for cross-omics transfer tasks." tags: [foundation-models, multi-omics-foundation-models, single-cell-foundation-models] tasks: [Foundation Model] modalities: [Multi-Omics, Single Cell] organism: [] api: false - id: scpair name: "scPair" type: model url: https://github.com/quon-titative-biology/scPair description: "Bidirectional feedforward network for single-cell multimodal analysis with cross-modality prediction leveraging single-cell atlases." tags: [foundation-models, multi-omics-foundation-models, single-cell-foundation-models] tasks: [Foundation Model] modalities: [Multi-Omics, Single Cell] organism: [] api: false - id: scprint name: "scPRINT" type: model url: https://github.com/cantinilab/scPRINT description: "Pretrained on 50M cells for scRNA-seq denoising & zero imputation." tags: [llm-for-biology] tasks: [Language Modeling] modalities: [Text] organism: [] api: false - id: sei name: "Sei" type: model url: https://github.com/FunctionLab/sei-framework description: "Sequence-to-function framework learning a genome-wide regulatory activity code from DNA sequences for variant effect prediction." tags: [foundation-models, genomics-foundation-models] tasks: [Foundation Model] modalities: [Genomics] organism: [] api: false - id: spatialglue name: "SpatialGlue" type: model url: https://github.com/zhanglabtools/SpatialGlue description: "Graph attention network for spatial multi-omics integration jointly embedding spatial transcriptomics with chromatin accessibility or proteomics." tags: [foundation-models, multi-omics-foundation-models, single-cell-foundation-models] tasks: [Foundation Model] modalities: [Multi-Omics, Single Cell] organism: [] api: false - id: targetdiff name: "TargetDiff" type: model url: https://github.com/guanjq/targetdiff description: "3D equivariant diffusion model for structure-based drug design." tags: [drug-discovery, molecular-generation] tasks: [Drug Discovery, Molecular Generation] modalities: [Small Molecule] organism: [] api: false - id: tgsa name: "TGSA" type: model url: https://github.com/violet-sto/TGSA description: "Tumor gene set and attention-based model leveraging biological pathway knowledge for drug response prediction." tags: [drug-discovery, drug-response-prediction] tasks: [Drug Discovery, Drug Response Prediction] modalities: [Small Molecule] organism: [] api: false - id: toad name: "TOAD" type: model url: https://github.com/mahmoodlab/TOAD description: "Tumor Origin Assessment via Deep-learning; weakly-supervised multi-task model predicting cancer primary origin from H&E whole-slide images." tags: [foundation-models, multi-modal-foundation-models] tasks: [Foundation Model] modalities: [Multi-Modal] organism: [] api: false - id: tosica name: "TOSICA" type: model url: https://github.com/JackieHanlaopo/TOSICA description: "Transformer-based framework for one-stop interpretable cell-type annotation supporting cross-dataset and cross-species transfer." tags: [domain-alignment, foundation-models, single-cell-foundation-models] tasks: [Domain Alignment, Foundation Model] modalities: [Single Cell] organism: [] api: false - id: totalvi name: "totalVI" type: model url: https://github.com/scverse/scvi-tools description: "Probabilistic framework for joint analysis of paired scRNA-seq and protein (CITE-seq) data enabling multi-modal cell state representation across single-cell datasets." tags: [foundation-models, multi-omics-foundation-models, single-cell-foundation-models] tasks: [Foundation Model] modalities: [Multi-Omics, Single Cell] organism: [] api: false - id: transformercpi name: "TransformerCPI" type: model url: https://github.com/lifanchen-simm/transformerCPI description: "CPI prediction using Transformer." tags: [compound-protein-interaction, drug-discovery] tasks: [Compound-Protein Interaction, Drug Discovery] modalities: [Protein, Small Molecule] organism: [] api: false - id: transigen name: "TranSiGen" type: model url: https://github.com/myzhengSIMM/TranSiGen description: "Dual-VAE architecture for ligand-based virtual screening, drug response prediction, and drug repurposing using chemical-induced transcriptional profiles." tags: [drug-discovery, drug-repurposing] tasks: [Drug Discovery, Drug Repurposing] modalities: [Small Molecule] organism: [] api: false - id: uce name: "UCE" type: model url: https://github.com/snap-stanford/UCE description: "Universal Cell Embeddings: zero-shot single-cell embedding model trained on 36M cells across species, tissues, and assays without fine-tuning." tags: [foundation-models, single-cell-foundation-models, transcriptomics-foundation-models] tasks: [Foundation Model] modalities: [Single Cell, Transcriptomics] organism: [] api: false - id: uni name: "UNI" type: model url: https://github.com/mahmoodlab/UNI description: "General-purpose self-supervised pathology foundation model trained on 100K+ whole-slide images for diverse computational pathology tasks." tags: [foundation-models, single-cell-foundation-models, spatial-foundation-models] tasks: [Foundation Model] modalities: [Single Cell, Spatial Transcriptomics] organism: [] api: false - id: uni_mol name: "Uni-Mol" type: model url: https://github.com/deepmodeling/Uni-Mol description: "3D molecular pretraining framework for universal representation learning on molecules and protein pockets." tags: [compound-embedding, compound-foundation-models, foundation-models] tasks: [Foundation Model] modalities: [Small Molecule] organism: [] api: false - id: unitednet name: "UnitedNet" type: model url: https://github.com/LiuLab-Bioelectronics-Harvard/UnitedNet description: "Interpretable multi-task deep neural network for single-cell multi-omics integration spanning transcriptomics, chromatin accessibility, and proteomics." tags: [foundation-models, multi-omics-foundation-models, single-cell-foundation-models] tasks: [Foundation Model] modalities: [Multi-Omics, Single Cell] organism: [] api: false - id: virchow name: "Virchow" type: model url: https://huggingface.co/paige-ai/Virchow description: "Million-slide digital pathology foundation model using a vision transformer and self-supervised distillation for tile-level pathology image representation." tags: [foundation-models, multi-modal-foundation-models] tasks: [Foundation Model] modalities: [Multi-Modal] organism: [] api: false - id: autozyme name: "AutoZyme" type: toolkit url: https://github.com/ElliotXie/autozyme description: "Autonomous agentic framework that speeds up bioinformatics software (e.g. Scanpy, Seurat) on CPUs while preserving the original results." tags: [preprocessing-tools] tasks: [Preprocessing] modalities: [] organism: [] api: false - id: biopython name: "Biopython" type: toolkit url: https://biopython.org/ description: "Collection of Python tools for biological computation including sequence analysis, structure parsing, and database access." tags: [preprocessing-tools] tasks: [Preprocessing] modalities: [] organism: [] api: false - id: casper name: "CaSpER" type: toolkit url: https://github.com/akdess/CaSpER description: "CNV identification and visualization by integrative analysis of single-cell or bulk RNA-seq data." tags: [preprocessing-tools] tasks: [Preprocessing] modalities: [] organism: [] api: false - id: cellcharter name: "CellCharter" type: toolkit url: https://github.com/CSOgroup/cellcharter description: "Identification and characterization of spatial cell niches from spatial transcriptomics using VAEs and Gaussian mixture models." tags: [preprocessing-tools] tasks: [Preprocessing] modalities: [] organism: [] api: false - id: cellchat name: "CellChat" type: toolkit url: https://github.com/sqjin/CellChat description: "Inference and analysis of cell-cell communication ligand-receptor networks from single-cell transcriptomics data." tags: [preprocessing-tools] tasks: [Preprocessing] modalities: [] organism: [] api: false - id: celltypist name: "CellTypist" type: toolkit url: https://github.com/Teichlab/celltypist description: "Automated cell type annotation for scRNA-seq." tags: [preprocessing-tools] tasks: [Preprocessing] modalities: [] organism: [] api: false - id: chatspatial name: "ChatSpatial" type: toolkit url: https://github.com/cafferychen777/ChatSpatial description: "MCP server for spatial transcriptomics analysis via natural language." tags: [preprocessing-tools] tasks: [Preprocessing] modalities: [] organism: [] api: false - id: chemistry_development_kit name: "Chemistry Development Kit" type: toolkit url: https://github.com/cdk/cdk description: "Cheminformatics software & machine learning tools." tags: [preprocessing-tools] tasks: [Preprocessing] modalities: [] organism: [] api: false - id: commot name: "COMMOT" type: toolkit url: https://github.com/zcang/COMMOT description: "Optimal transport-based framework for screening cell-cell communication in spatial transcriptomics." tags: [preprocessing-tools] tasks: [Preprocessing] modalities: [] organism: [] api: false - id: deepchem name: "DeepChem" type: toolkit url: https://github.com/deepchem/deepchem description: "Deep learning library for drug discovery, quantum chemistry, and materials science." tags: [preprocessing-tools] tasks: [Preprocessing] modalities: [] organism: [] api: false - id: deeptalk name: "DeepTalk" type: toolkit url: https://github.com/JiangBioLab/DeepTalk description: "Graph attention network for deciphering cell-cell communication from spatial transcriptomics." tags: [preprocessing-tools] tasks: [Preprocessing] modalities: [] organism: [] api: false - id: doubletfinder name: "DoubletFinder" type: toolkit url: https://github.com/chris-mcginnis-ucsf/DoubletFinder description: "Machine learning approach for detecting multiplet (doublet) artifacts in single-cell RNA-seq data." tags: [preprocessing-tools] tasks: [Preprocessing] modalities: [] organism: [] api: false - id: flashdeconv name: "FlashDeconv" type: toolkit url: https://github.com/cafferychen777/flashdeconv description: "High-performance spatial transcriptomics deconvolution (~1M spots in ~3 min)." tags: [preprocessing-tools] tasks: [Preprocessing] modalities: [] organism: [] api: false - id: gromacs name: "GROMACS" type: toolkit url: https://www.gromacs.org/ description: "Molecular dynamics simulation package for biochemical molecules." tags: [preprocessing-tools] tasks: [Preprocessing] modalities: [] organism: [] api: false - id: harmony name: "Harmony" type: toolkit url: https://github.com/immunogenomics/harmony description: "Fast and scalable integration of single-cell data across datasets, conditions, technologies, and species." tags: [preprocessing-tools] tasks: [Preprocessing] modalities: [] organism: [] api: false - id: kallisto name: "kallisto" type: toolkit url: https://pachterlab.github.io/kallisto/ description: "Near-optimal RNA-seq quantification using pseudoalignment for fast transcript abundance estimation." tags: [preprocessing-tools] tasks: [Preprocessing] modalities: [] organism: [] api: false - id: linger name: "LINGER" type: toolkit url: https://github.com/Durenlab/LINGER description: "Neural network for gene regulatory network inference from single-cell multiome (RNA+ATAC-seq) data with bulk data pretraining." tags: [preprocessing-tools] tasks: [Preprocessing] modalities: [] organism: [] api: false - id: mdanalysis name: "MDAnalysis" type: toolkit url: https://www.mdanalysis.org/ description: "Python library for analyzing and altering molecular dynamics simulation trajectories." tags: [preprocessing-tools] tasks: [Preprocessing] modalities: [] organism: [] api: false - id: mogonet name: "MOGONET" type: toolkit url: https://github.com/txWang/MOGONET description: "Multi-omics graph convolutional network framework for patient classification and biomarker identification." tags: [preprocessing-tools] tasks: [Preprocessing] modalities: [] organism: [] api: false - id: monocle3 name: "Monocle3" type: toolkit url: https://cole-trapnell-lab.github.io/monocle3/ description: "Single-cell trajectory analysis tool for learning developmental trajectories and ordering cells in pseudotime." tags: [preprocessing-tools] tasks: [Preprocessing] modalities: [] organism: [] api: false - id: ncem name: "NCEM" type: toolkit url: https://github.com/theislab/ncem description: "GNN-based model for learning intercellular communication from spatial graphs of cells." tags: [preprocessing-tools] tasks: [Preprocessing] modalities: [] organism: [] api: false - id: numbat name: "Numbat" type: toolkit url: https://github.com/kharchenkolab/numbat description: "Haplotype-aware copy number variation inference from single-cell RNA-seq using hidden Markov models." tags: [preprocessing-tools] tasks: [Preprocessing] modalities: [] organism: [] api: false - id: openmm name: "OpenMM" type: toolkit url: https://openmm.org/ description: "High-performance toolkit for molecular simulation and GPU-accelerated MD." tags: [preprocessing-tools] tasks: [Preprocessing] modalities: [] organism: [] api: false - id: rdkit name: "RDKit" type: toolkit url: https://github.com/rdkit/rdkit description: "Cheminformatics software & machine learning toolkit." tags: [preprocessing-tools] tasks: [Preprocessing] modalities: [] organism: [] api: false - id: scanpy name: "Scanpy" type: toolkit url: https://scanpy.readthedocs.io/en/stable/ description: "Python library for scRNA-seq analysis." tags: [preprocessing-tools] tasks: [Preprocessing] modalities: [] organism: [] api: false - id: scenic name: "SCENIC" type: toolkit url: https://github.com/aertslab/SCENIC description: "Single-cell regulatory network inference and clustering linking transcription factors to co-expressed gene modules." tags: [preprocessing-tools] tasks: [Preprocessing] modalities: [] organism: [] api: false - id: scipenn name: "sciPENN" type: toolkit url: https://github.com/jlakkis/sciPENN description: "RNN-based method for simultaneous protein expression prediction, uncertainty estimation, and cell-type label transfer from CITE-seq and scRNA-seq data." tags: [preprocessing-tools] tasks: [Preprocessing] modalities: [] organism: [] api: false - id: scvelo name: "scVelo" type: toolkit url: https://github.com/theislab/scvelo description: "RNA velocity estimation for single-cell transcriptomics, inferring the direction and speed of cell differentiation." tags: [preprocessing-tools] tasks: [Preprocessing] modalities: [] organism: [] api: false - id: scvi_tools name: "scvi-tools" type: toolkit url: https://scvi-tools.org/ description: "Probabilistic models for single-cell omics data analysis." tags: [preprocessing-tools] tasks: [Preprocessing] modalities: [] organism: [] api: false - id: seqbench name: "SeqBench" type: toolkit url: https://seqbench.com/ description: "Web-based molecular biology sequence workbench for primer design, cloning simulation (Gibson, Golden Gate, restriction digest), CRISPR guide RNA design, and sequence analysis, with a public REST API, OpenAPI 3.1 spec, and MCP server." tags: [preprocessing-tools] tasks: [Preprocessing] modalities: [] organism: [] api: false - id: seurat name: "Seurat" type: toolkit url: https://satijalab.org/seurat/ description: "R library for scRNA-seq analysis." tags: [preprocessing-tools] tasks: [Preprocessing] modalities: [] organism: [] api: false - id: squidpy name: "Squidpy" type: toolkit url: https://squidpy.readthedocs.io/ description: "Python library for spatial single-cell analysis." tags: [preprocessing-tools] tasks: [Preprocessing] modalities: [] organism: [] api: false - id: stagate name: "STAGATE" type: toolkit url: https://github.com/RucDongLab/STAGATE description: "Adaptive graph attention auto-encoder for spatial domain identification in spatial transcriptomics." tags: [preprocessing-tools] tasks: [Preprocessing] modalities: [] organism: [] api: false - id: star name: "STAR" type: toolkit url: https://github.com/alexdobin/STAR description: "Ultrafast universal RNA-seq aligner with support for spliced alignment and single-cell quantification via STARsolo." tags: [preprocessing-tools] tasks: [Preprocessing] modalities: [] organism: [] api: false - id: tigon name: "TIGON" type: toolkit url: https://github.com/yutongo/TIGON description: "Neural optimal transport method for reconstructing growth and dynamic trajectories from single-cell transcriptomics." tags: [preprocessing-tools] tasks: [Preprocessing] modalities: [] organism: [] api: false