{"id":"2ome-lm-2025","kind":"source","name":"2OMe-LM: predicting 2′-O-methylation sites in human RNA using a pre-trained RNA language model","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["rna-transcriptomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12342186/","version":"journal full text in PMC","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.1093/bioinformatics/btaf417","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"54fe4db6f35c03d0d4f3ef4da720eb26a832199372c56d0956609ff07af750ee","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12342186/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:38.332Z","legacy_paper":{"id":"2ome-lm-2025","title":"2OMe-LM: predicting 2′-O-methylation sites in human RNA using a pre-trained RNA language model","year":2025,"publication_status":"peer_reviewed","version":"journal full text in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12342186/","primary_domain":"rna-transcriptomes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.1093/bioinformatics/btaf417","notes":"Numeric result checked against Table 1. in primary full-text XML; journal/source: Bioinformatics."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"adar-gpt-editing-2026","kind":"source","name":"ADAR-GPT: A continually fine-tuned language model for predicting A-to-I RNA editing sites","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["rna-transcriptomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12798952/","version":"version of record","retrieved_at":"2026-09-15T23:37:05Z","doi":"10.1073/pnas.2529073123","publication_status":"peer_reviewed","year":2026,"artifact_sha256":"cc8c7eb928f246f1f347a8822f614cd3475381c35eef6d579032ce441580198e","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12798952/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:06Z","legacy_paper":{"id":"adar-gpt-editing-2026","title":"ADAR-GPT: A continually fine-tuned language model for predicting A-to-I RNA editing sites","year":2026,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12798952/","primary_domain":"rna-transcriptomes","retrieved_utc":"2026-09-15T23:37:05Z","notes":"Primary full text verified via Europe PMC fullTextXML; venue: Proceedings of the National Academy of Sciences of the United States of America; PMC ID: PMC12798952. RNA editing site benchmark on a restricted liver validation set.","doi":"10.1073/pnas.2529073123"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"akscore-2020","kind":"source","name":"AK-Score: Accurate Protein-Ligand Binding Affinity Prediction Using an Ensemble of 3D-Convolutional Neural Networks","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC7697539/","version":"version of record","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.3390/ijms21228424","publication_status":"peer_reviewed","year":2020,"artifact_sha256":"40cfd28dcd587599768ec99a6590ec593486475ff01c7b1d1f229b44aa91bf8d","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC7697539/fullTextXML","artifact_retrieved_at":"2026-09-16T10:44:03.436853+00:00","legacy_paper":{"id":"akscore-2020","title":"AK-Score: Accurate Protein-Ligand Binding Affinity Prediction Using an Ensemble of 3D-Convolutional Neural Networks","year":2020,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC7697539/","primary_domain":"molecular-interactions","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: International Journal of Molecular Sciences; PMC ID: PMC7697539.","doi":"10.3390/ijms21228424"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"antibody-deamidation-plm-2024","kind":"source","name":"The Accurate Prediction of Antibody Deamidations by Combining High-Throughput Automated Peptide Mapping and Protein Language Model-Based Deep Learning","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["proteins-complexes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11417914/","version":"journal full text in PMC","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.3390/antib13030074","publication_status":"peer_reviewed","year":2024,"artifact_sha256":"aa049f6d78e29540ba902a0d3b7f53d49e9833e4dad9869f79ac27664e8c150b","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11417914/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:38.478Z","legacy_paper":{"id":"antibody-deamidation-plm-2024","title":"The Accurate Prediction of Antibody Deamidations by Combining High-Throughput Automated Peptide Mapping and Protein Language Model-Based Deep Learning","year":2024,"publication_status":"peer_reviewed","version":"journal full text in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11417914/","primary_domain":"proteins-complexes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.3390/antib13030074","notes":"Numeric result checked against Table 1 in primary full-text XML; journal/source: Antibodies."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"antibody-flexibility-2025","kind":"source","name":"Enhancing antibody-antigen interaction prediction with atomic flexibility","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12530544/","version":"version of record","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.1371/journal.pcbi.1013576","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"57e64694c69052ed0495570e12ebfb4bb6c0ad152219f23827cd4b1cb53450ef","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12530544/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:55.400Z","legacy_paper":{"id":"antibody-flexibility-2025","title":"Enhancing antibody-antigen interaction prediction with atomic flexibility","year":2025,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12530544/","primary_domain":"molecular-interactions","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: PLOS Computational Biology; PMC ID: PMC12530544.","doi":"10.1371/journal.pcbi.1013576"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"arsenal-regulatory-dna-2026","kind":"source","name":"Short-Context Regulatory DNA Language Models with Motif-Discovery Regularization","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["dna-genomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12889687/","version":"preprint version in PMC","retrieved_at":"2026-09-15T23:37:05Z","doi":"10.64898/2026.02.05.703637","publication_status":"preprint","year":2026,"artifact_sha256":"4a264956e47fc633aaff6573aac368dc691dd5de709b27c7421c078608ff542a","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12889687/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:06Z","legacy_paper":{"id":"arsenal-regulatory-dna-2026","title":"Short-Context Regulatory DNA Language Models with Motif-Discovery Regularization","year":2026,"publication_status":"preprint","version":"preprint version in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12889687/","primary_domain":"dna-genomes","retrieved_utc":"2026-09-15T23:37:05Z","notes":"Primary full text verified via Europe PMC fullTextXML; venue: bioRxiv; PMC ID: PMC12889687. Preprint; result is a supervised downstream model rather than a general-purpose DNA foundation model.","doi":"10.64898/2026.02.05.703637"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"b2-2ome-lm-2025","kind":"result","name":"2OMe-LM · AUC · human RNA 2OMe sites","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["human RNA 2-prime-O-methylation site prediction"]},"source_ids":["2ome-lm-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-2ome-lm-2025"}],"attributes":{"printed_value":"0.919","numeric_value":"0.919","metric":"AUC","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 1, 2OMe-LM row, AUC column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:38.332Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 1, 2OMe-LM row, AUC column; cell: 0.919","artifact_sha256":"54fe4db6f35c03d0d4f3ef4da720eb26a832199372c56d0956609ff07af750ee","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12342186/fullTextXML"},"legacy_id":"b2-2ome-lm-2025","legacy_row":{"id":"b2-2ome-lm-2025","paper_id":"2ome-lm-2025","domain_id":"rna-transcriptomes","task":"human RNA 2-prime-O-methylation site prediction","model":"2OMe-LM","model_version":"not stated in table","dataset":"human RNA 2OMe sites","dataset_version":"","split":"5-fold cross-validation","metric":"AUC","value":"0.919","unit":"fraction","uncertainty":"","protocol":"pretrained RNA language model predictor","source_locator":"Table 1, 2OMe-LM row, AUC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12342186/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-antibody-deamidation-plm-2024","kind":"result","name":"ESM-2 650M embeddings + classifier · accuracy · antibody peptide-mapping training dataset","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["antibody deamidation-site prediction"]},"source_ids":["antibody-deamidation-plm-2024"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-antibody-deamidation-plm-2024"}],"attributes":{"printed_value":"0.944","numeric_value":"0.944","metric":"accuracy","metric_direction":"unknown","unit":"fraction","uncertainty":"± 0.012","source_locator":"Table 1, Global embeddings only row, Accuracy column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:38.478Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 1, Global embeddings only row, Accuracy column; cell: 0.944 ± 0.012","artifact_sha256":"aa049f6d78e29540ba902a0d3b7f53d49e9833e4dad9869f79ac27664e8c150b","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11417914/fullTextXML"},"legacy_id":"b2-antibody-deamidation-plm-2024","legacy_row":{"id":"b2-antibody-deamidation-plm-2024","paper_id":"antibody-deamidation-plm-2024","domain_id":"proteins-complexes","task":"antibody deamidation-site prediction","model":"ESM-2 650M embeddings + classifier","model_version":"esm2_t33_650m_UR50D","dataset":"antibody peptide-mapping training dataset","dataset_version":"","split":"fivefold stratified CV","metric":"accuracy","value":"0.944","unit":"fraction","uncertainty":"± 0.012","protocol":"global contextual embeddings only","source_locator":"Table 1, Global embeddings only row, Accuracy column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11417914/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"b2-barcodebert-2026","kind":"result","name":"BarcodeBERT (4–4-4) · accuracy · DNA barcodes of unseen species","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["unseen-species genus classification"]},"source_ids":["barcodebert-2026"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-barcodebert-2026"}],"attributes":{"printed_value":"78.5","numeric_value":"78.5","metric":"accuracy","metric_direction":"unknown","unit":"percent","uncertainty":null,"source_locator":"Table 1, BarcodeBERT (4–4-4) row, unseen-species genus-level 1-NN Acc (%) column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558051+00:00","notes":"Resolved the two-level column header: Acc (%) falls under genus-level 1-NN probe of unseen species, not seen-species classification or BIN reconstruction. BarcodeBERT (4–4-4) has 78.5 in this cell.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 78.5.","artifact_sha256":"493f9fe70b483780ba76d51ccf217d3ca83539c82b89917fd3ccebe2b6eb831d","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13008329/fullTextXML"},"legacy_id":"b2-barcodebert-2026","legacy_row":{"id":"b2-barcodebert-2026","paper_id":"barcodebert-2026","domain_id":"dna-genomes","task":"unseen-species genus classification","model":"BarcodeBERT (4–4-4)","model_version":"4–4–4","dataset":"DNA barcodes of unseen species","dataset_version":"","split":"1-NN probe","metric":"accuracy","value":"78.5","unit":"percent","uncertainty":"","protocol":"genus-level nearest-neighbor probe on species unseen in training","source_locator":"Table 1, BarcodeBERT (4–4-4) row, unseen-species genus-level 1-NN Acc (%) column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13008329/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-birna-bert-2025","kind":"result","name":"BiRNA-BERT · F1 · extremely long-sequence species classification","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["extremely long RNA species classification"]},"source_ids":["birna-bert-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-birna-bert-2025"}],"attributes":{"printed_value":"0.804","numeric_value":"0.804","metric":"F1","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 2, BiRNA-BERT row, F1 Score column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:38.292Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 2, BiRNA-BERT row, F1 Score column; cell: 0.804","artifact_sha256":"bf7dbc52b6515301c77010c513f13e676c38395ddc82c20310171f518690c152","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12635123/fullTextXML"},"legacy_id":"b2-birna-bert-2025","legacy_row":{"id":"b2-birna-bert-2025","paper_id":"birna-bert-2025","domain_id":"rna-transcriptomes","task":"extremely long RNA species classification","model":"BiRNA-BERT","model_version":"not stated in table","dataset":"extremely long-sequence species classification","dataset_version":"","split":"paper evaluation","metric":"F1","value":"0.804","unit":"fraction","uncertainty":"","protocol":"adaptive tokenization on full-length long RNA sequences","source_locator":"Table 2, BiRNA-BERT row, F1 Score column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12635123/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-cathe2-2025","kind":"result","name":"CATHe2 + ProstT5 · F1 · CATH superfamily benchmark","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["CATH superfamily annotation"]},"source_ids":["cathe2-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-cathe2-2025"}],"attributes":{"printed_value":"82.3","numeric_value":"82.3","metric":"F1","metric_direction":"unknown","unit":"percent","uncertainty":"± 1.3 percentage points","source_locator":"Table 3, ProstT5 full row, F1 score column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:38.366Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 3, ProstT5 full row, F1 score column; cell: 82.3% ± 1.3%","artifact_sha256":"713dbfb6ec1cc1aa85c0543eb93aafa0b45b8873df28053b765dd0a1b6d9b563","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12631783/fullTextXML"},"legacy_id":"b2-cathe2-2025","legacy_row":{"id":"b2-cathe2-2025","paper_id":"cathe2-2025","domain_id":"proteins-complexes","task":"CATH superfamily annotation","model":"CATHe2 + ProstT5","model_version":"full ProstT5","dataset":"CATH superfamily benchmark","dataset_version":"","split":"paper evaluation","metric":"F1","value":"82.3","unit":"percent","uncertainty":"± 1.3 percentage points","protocol":"amino-acid and structural alphabet embedding classifier","source_locator":"Table 3, ProstT5 full row, F1 score column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12631783/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"b2-clathrin-plm-2025","kind":"result","name":"ESM-2 embedding + paper classifier · accuracy · CLA-IND0.6","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["clathrin protein classification"]},"source_ids":["clathrin-plm-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-clathrin-plm-2025"}],"attributes":{"printed_value":"0.916","numeric_value":"0.916","metric":"accuracy","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 2, Independent test / ESM-2 row, ACC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558194+00:00","notes":"Resolved the blank evaluation-strategy cells by their independent-test row group. ESM-2 ACC is 0.916 there; the cross-validation ESM-2 ACC is instead 0.873. This is the paper classifier using embeddings, not a standalone checkpoint.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 0.916.","artifact_sha256":"2edc86b25707c1b737d26117093ce8d856e79cc5d0b335f27c1c341f887f1c7e","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12238356/fullTextXML"},"legacy_id":"b2-clathrin-plm-2025","legacy_row":{"id":"b2-clathrin-plm-2025","paper_id":"clathrin-plm-2025","domain_id":"proteins-complexes","task":"clathrin protein classification","model":"ESM-2 embedding + paper classifier","model_version":"not stated in table","dataset":"CLA-IND0.6","dataset_version":"","split":"independent test","metric":"accuracy","value":"0.916","unit":"fraction","uncertainty":"","protocol":"single-feature ESM-2 embedding comparison","source_locator":"Table 2, Independent test / ESM-2 row, ACC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12238356/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-cobra-rna-binding-2026","kind":"result","name":"ERNIE-RNA + CoBRA · MCC · CoBRA compound-binding test set","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA compound-binding site prediction"]},"source_ids":["cobra-rna-binding-2026"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-cobra-rna-binding-2026"}],"attributes":{"printed_value":"0.657","numeric_value":"0.657","metric":"MCC","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 2, ERNIE-RNA / TCL focal row, MCC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558197+00:00","notes":"Matched ERNIE-RNA jointly with TCL focal loss, then the MCC column. Table 2 explicitly reports test-set models. The cell is 0.657, distinct from AUROC 0.868.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 0.657.","artifact_sha256":"8c6a6f00f5fa5f62acf301a66e9e6fa9ef11c7a05ad9b7447d2ade2ce8eba793","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12790621/fullTextXML"},"legacy_id":"b2-cobra-rna-binding-2026","legacy_row":{"id":"b2-cobra-rna-binding-2026","paper_id":"cobra-rna-binding-2026","domain_id":"rna-transcriptomes","task":"RNA compound-binding site prediction","model":"ERNIE-RNA + CoBRA","model_version":"not stated in table","dataset":"CoBRA compound-binding test set","dataset_version":"","split":"test set","metric":"MCC","value":"0.657","unit":"unitless","uncertainty":"","protocol":"ERNIE-RNA embedding with TCL focal loss","source_locator":"Table 2, ERNIE-RNA / TCL focal row, MCC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12790621/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-codonbert-vaccines-2024","kind":"result","name":"CodonBERT · Spearman rho · flu-vaccine sequences","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["flu-vaccine mRNA property prediction"]},"source_ids":["codonbert-vaccines-2024"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-codonbert-vaccines-2024"}],"attributes":{"printed_value":"0.81","numeric_value":"0.81","metric":"Spearman rho","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 2, CodonBERT row, Flu vaccines Spearman correlation column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558201+00:00","notes":"Matched the CodonBERT row and Flu vaccines column (0.81). The table footnote identifies regression columns as Spearman rank correlation and singles out E. coli as classification; this is not a flu-vaccine accuracy score.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 0.81.","artifact_sha256":"2968073753e6d44feff9c08b131edf23145e95b171434539dddf77bb92847033","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11368176/fullTextXML"},"legacy_id":"b2-codonbert-vaccines-2024","legacy_row":{"id":"b2-codonbert-vaccines-2024","paper_id":"codonbert-vaccines-2024","domain_id":"rna-transcriptomes","task":"flu-vaccine mRNA property prediction","model":"CodonBERT","model_version":"not stated in table","dataset":"flu-vaccine sequences","dataset_version":"","split":"paper evaluation","metric":"Spearman rho","value":"0.81","unit":"unitless","uncertainty":"","protocol":"codon-based model fine-tuned for downstream regression","source_locator":"Table 2, CodonBERT row, Flu vaccines Spearman correlation column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11368176/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-dart-eval-regulatory-2024","kind":"result","name":"DNABERT-2 · accuracy · DART-Eval cCREs versus matched shuffled controls","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["regulatory element identification"]},"source_ids":["dart-eval-regulatory-2024"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-dart-eval-regulatory-2024"}],"attributes":{"printed_value":"0.876","numeric_value":"0.876","metric":"accuracy","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 3 (PDF page 5), DNABERT-2 row, Zero-Shot Accuracy column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558203+00:00","notes":"Inspected the pinned NeurIPS primary PDF table and explanatory text. The DNABERT-2 row reports 0.876 under Zero-Shot Accuracy. The caption defines this as pairwise prioritization of positives over matched controls, distinct from supervised absolute accuracy.","evidence":"DNABERT-2; zero-shot accuracy 0.876; probed absolute/paired 0.847/0.943; fine-tuned absolute/paired 0.913/0.973.","artifact_sha256":"e5aee5b1f7cc6fd961b1d2a131d02cf243b79e091d5e418fbabee7fde9b39b22","retrieval_url":"https://proceedings.neurips.cc/paper_files/paper/2024/file/71998bfc3217ffe1cca1ee084dfadadd-Paper-Datasets_and_Benchmarks_Track.pdf"},"legacy_id":"b2-dart-eval-regulatory-2024","legacy_row":{"id":"b2-dart-eval-regulatory-2024","paper_id":"dart-eval-regulatory-2024","domain_id":"dna-genomes","task":"regulatory element identification","model":"DNABERT-2","model_version":"not stated in table","dataset":"DART-Eval cCREs versus matched shuffled controls","dataset_version":"","split":"paper evaluation","metric":"accuracy","value":"0.876","unit":"fraction","uncertainty":"","protocol":"zero-shot likelihood ranking: higher likelihood for cCRE than matched control","source_locator":"Table 3, DNABERT-2 row, Zero-Shot Accuracy column","source_url":"https://proceedings.neurips.cc/paper_files/paper/2024/file/71998bfc3217ffe1cca1ee084dfadadd-Paper-Datasets_and_Benchmarks_Track.pdf","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-dnabert2-enhancer-2025","kind":"result","name":"DNABERT2-Enhancer · AUC · Liu training dataset","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["enhancer recognition"]},"source_ids":["dnabert2-enhancer-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-dnabert2-enhancer-2025"}],"attributes":{"printed_value":"0.965","numeric_value":"0.965","metric":"AUC","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 4, first-layer DNABERT2-Enhancer row, AUC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558204+00:00","notes":"Resolved the first-layer row group. DNABERT2-Enhancer AUC is 0.965, whereas second-layer AUC is 0.933. The caption explicitly describes 5-fold cross-validation on Liu training data, not an independent held-out test.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 0.965.","artifact_sha256":"d052b80efe7bfc1380994ad28503a5575f04ef940f74d5c9c137cb4ba6827863","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11981215/fullTextXML"},"legacy_id":"b2-dnabert2-enhancer-2025","legacy_row":{"id":"b2-dnabert2-enhancer-2025","paper_id":"dnabert2-enhancer-2025","domain_id":"dna-genomes","task":"enhancer recognition","model":"DNABERT2-Enhancer","model_version":"not stated in table","dataset":"Liu training dataset","dataset_version":"","split":"5-fold cross-validation","metric":"AUC","value":"0.965","unit":"fraction","uncertainty":"","protocol":"first-layer enhancer versus non-enhancer classifier","source_locator":"Table 4, first-layer DNABERT2-Enhancer row, AUC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11981215/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-eden-genomic-classification-2026","kind":"result","name":"DNABERT-2 · MCC · GUE H-CPD","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["human core-promoter classification"]},"source_ids":["eden-genomic-classification-2026"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-eden-genomic-classification-2026"}],"attributes":{"printed_value":"70.52","numeric_value":"70.52","metric":"MCC","metric_direction":"unknown","unit":"percent","uncertainty":null,"source_locator":"Table 5, DNABERT-2 row, H-CPD (MCC) column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:37.531Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 5, DNABERT-2 row, H-CPD (MCC) column; cell: 70.52","artifact_sha256":"38a6e26b3caffe8e021a2b0b672218e783aca9ee42046765e323946813015e65","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12879454/fullTextXML"},"legacy_id":"b2-eden-genomic-classification-2026","legacy_row":{"id":"b2-eden-genomic-classification-2026","paper_id":"eden-genomic-classification-2026","domain_id":"dna-genomes","task":"human core-promoter classification","model":"DNABERT-2","model_version":"not stated in table","dataset":"GUE H-CPD","dataset_version":"","split":"paper evaluation","metric":"MCC","value":"70.52","unit":"percent","uncertainty":"","protocol":"DNABERT-2 comparator in consolidated H-CPD table; rerun provenance not explicit","source_locator":"Table 5, DNABERT-2 row, H-CPD (MCC) column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12879454/","evaluation_origin":"paper_compilation","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-ernie-rna-2025","kind":"result","name":"ERNIE-RNA · binary F1 · bpRNA-new","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary-structure prediction"]},"source_ids":["ernie-rna-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-ernie-rna-2025"}],"attributes":{"printed_value":"0.575","numeric_value":"0.575","metric":"binary F1","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 2, ERNIE-RNA zero-shot row, bpRNA-new F1-Score (binary) column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558206+00:00","notes":"Resolved bpRNA-new as the first three-column dataset group and F1-Score (binary) as its third metric. ERNIE-RNA zero shot is 86M and reports 0.575; RNA3DB-2D F1 is instead 0.542.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 0.575.","artifact_sha256":"0bd1d4b3cbf5d59d452cec4864614947861efcee050ba07e7de395cd90630047","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12627772/fullTextXML"},"legacy_id":"b2-ernie-rna-2025","legacy_row":{"id":"b2-ernie-rna-2025","paper_id":"ernie-rna-2025","domain_id":"rna-transcriptomes","task":"RNA secondary-structure prediction","model":"ERNIE-RNA","model_version":"86M","dataset":"bpRNA-new","dataset_version":"","split":"cross-family test","metric":"binary F1","value":"0.575","unit":"fraction","uncertainty":"","protocol":"zero-shot attention-derived base-pair prediction","source_locator":"Table 2, ERNIE-RNA zero-shot row, bpRNA-new F1-Score (binary) column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12627772/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-esm2-ofs-fitness-2025","kind":"result","name":"ESM2 OFS pseudo-perplexity · Spearman rho · ProteinGym substitutions","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["protein variant fitness prediction"]},"source_ids":["esm2-ofs-fitness-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-esm2-ofs-fitness-2025"}],"attributes":{"printed_value":"0.403","numeric_value":"0.403","metric":"Spearman rho","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Published PDF page 6 (033014-6), Table I, ESM2: OFS PP row, Aggregate mean column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source review, not human review","reviewed_at":"2026-09-16T10:45:41.099916+00:00","notes":"Publisher PDF retrieved through official APS harvest endpoint after direct download returned403. Table I is ProteinGym substitutions, not indels TableII. Last column aggregate mean0.403; separate function categories precede it. This verifies reported score, not experimental reproduction. Comparator rows in this table are sourced from ProteinGym; OFS PP is authors own method.","evidence":"Headers: Activity(43), Binding(14), Expression(17), Organismal fitness(77), Stability(66), Aggregate mean. ESM2:OFS PP row:0.393,0.279,0.397,0.331,0.507,0.403. Verified publisher PDF layout extraction against web-rendered primary PDF table.","artifact_sha256":"085ef646f11b8e5335c4b3d86b15fb6c7bf5edf4a80a8b622753ac69d9991a67","retrieval_url":"https://harvest.aps.org/v2/journals/articles/10.1103/zhx7-hcmm/fulltext"},"legacy_id":"b2-esm2-ofs-fitness-2025","legacy_row":{"id":"b2-esm2-ofs-fitness-2025","paper_id":"esm2-ofs-fitness-2025","domain_id":"proteins-complexes","task":"protein variant fitness prediction","model":"ESM2 OFS pseudo-perplexity","model_version":"not stated in table","dataset":"ProteinGym substitutions","dataset_version":"","split":"aggregate across assays","metric":"Spearman rho","value":"0.403","unit":"unitless","uncertainty":"","protocol":"authors’ zero-shot ESM2 OFS pseudo-perplexity evaluation; aggregate mean across ProteinGym substitution assays","source_locator":"Table I, ESM2: OFS PP row, Aggregate Mean Spearman correlation column","source_url":"https://journals.aps.org/prxlife/pdf/10.1103/zhx7-hcmm","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-fusion-breakpoint-foundation-models-2026","kind":"result","name":"Nucleotide Transformer + NN (middle) · ROC AUC · gene fusion breakpoint DNA sequences","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["gene fusion breakpoint classification"]},"source_ids":["fusion-breakpoint-foundation-models-2026"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-fusion-breakpoint-foundation-models-2026"}],"attributes":{"printed_value":"0.994","numeric_value":"0.994","metric":"ROC AUC","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 2, NT / NN (middle) row, ROC AUC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558209+00:00","notes":"Matched NT jointly with NN (middle) and ROC AUC 0.994 in the full-test-set table. NT with SVM reports 0.995 and is a separate pipeline.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 0.994.","artifact_sha256":"0f4d9de77f1e39cfd2164a20653d86370767da684dc22d17e09f589761abeb5f","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13182013/fullTextXML"},"legacy_id":"b2-fusion-breakpoint-foundation-models-2026","legacy_row":{"id":"b2-fusion-breakpoint-foundation-models-2026","paper_id":"fusion-breakpoint-foundation-models-2026","domain_id":"dna-genomes","task":"gene fusion breakpoint classification","model":"Nucleotide Transformer + NN (middle)","model_version":"not stated in table","dataset":"gene fusion breakpoint DNA sequences","dataset_version":"","split":"full test set","metric":"ROC AUC","value":"0.994","unit":"fraction","uncertainty":"","protocol":"middle embedding with neural-network classifier","source_locator":"Table 2, NT / NN (middle) row, ROC AUC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13182013/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-genomic-tokenizer-selection-2025","kind":"result","name":"Caduceus (character tokens) · MCC · genomic benchmark categories","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["regulatory sequence classification"]},"source_ids":["genomic-tokenizer-selection-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-genomic-tokenizer-selection-2025"}],"attributes":{"printed_value":"0.778","numeric_value":"0.778","metric":"MCC","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 2, Regulatory row, Caduceus (char) MCC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558210+00:00","notes":"Matched Regulatory row with Caduceus (char) column, 0.778. Caption establishes these as MCC summaries by category; model-size row identifies 3.9M parameters. This is an aggregated category result, not a single unspecified split.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 0.778.","artifact_sha256":"0a01c36fdd63f3f6db509777e61c3f87e8a298c810f8aef7974915aaa0655342","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12453675/fullTextXML"},"legacy_id":"b2-genomic-tokenizer-selection-2025","legacy_row":{"id":"b2-genomic-tokenizer-selection-2025","paper_id":"genomic-tokenizer-selection-2025","domain_id":"dna-genomes","task":"regulatory sequence classification","model":"Caduceus (character tokens)","model_version":"3.9M parameter variant","dataset":"genomic benchmark categories","dataset_version":"","split":"paper benchmark summary","metric":"MCC","value":"0.778","unit":"unitless","uncertainty":"","protocol":"task-category MCC across benchmark datasets","source_locator":"Table 2, Regulatory row, Caduceus (char) MCC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12453675/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-gsmformer-ppi-2026","kind":"result","name":"GSMFormer-PPI + ProstT5 · AUROC · paper PPI test set","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["protein-protein interaction prediction"]},"source_ids":["gsmformer-ppi-2026"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-gsmformer-ppi-2026"}],"attributes":{"printed_value":"0.988","numeric_value":"0.988","metric":"AUROC","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 6, ProstT5 embedding row, AUROC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558212+00:00","notes":"Matched ProstT5 embedding row and AUROC column, 0.988. Caption explicitly describes GSMFormer-PPI using embeddings as node features, not standalone ProstT5 prediction.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 0.988.","artifact_sha256":"9b364b5d73d16f2787f93f78f17dbe98b954ab9c2c64c1df960eec2e615eb3b4","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12873117/fullTextXML"},"legacy_id":"b2-gsmformer-ppi-2026","legacy_row":{"id":"b2-gsmformer-ppi-2026","paper_id":"gsmformer-ppi-2026","domain_id":"proteins-complexes","task":"protein-protein interaction prediction","model":"GSMFormer-PPI + ProstT5","model_version":"not stated in table","dataset":"paper PPI test set","dataset_version":"","split":"test set","metric":"AUROC","value":"0.988","unit":"fraction","uncertainty":"","protocol":"ProstT5 embeddings as graph node features","source_locator":"Table 6, ProstT5 embedding row, AUROC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12873117/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-megsite-2025","kind":"result","name":"MegSite + ESM3 · AUC · DNA-129_Test","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["DNA-binding residue prediction"]},"source_ids":["megsite-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-megsite-2025"}],"attributes":{"printed_value":"0.948","numeric_value":"0.948","metric":"AUC","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 2, DNA-129_Test / ESM3 row, AUC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558213+00:00","notes":"Resolved DNA-129_Test row group and ESM3 row. AUC is 0.948; the next numeric cell 0.582 is AP. Caption states an embedding comparison within MegSite.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 0.948.","artifact_sha256":"10d13122331813243d83b84fe6f9294eac7e7c03cde082ebed276191ac41089c","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12496013/fullTextXML"},"legacy_id":"b2-megsite-2025","legacy_row":{"id":"b2-megsite-2025","paper_id":"megsite-2025","domain_id":"proteins-complexes","task":"DNA-binding residue prediction","model":"MegSite + ESM3","model_version":"not stated in table","dataset":"DNA-129_Test","dataset_version":"","split":"independent test","metric":"AUC","value":"0.948","unit":"fraction","uncertainty":"","protocol":"ESM3 multimodal embedding ablation in MegSite","source_locator":"Table 2, DNA-129_Test / ESM3 row, AUC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12496013/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-mrna-lm-2025","kind":"result","name":"mRNA-LM · Spearman rho · mRNA half-life","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["mRNA half-life prediction"]},"source_ids":["mrna-lm-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-mrna-lm-2025"}],"attributes":{"printed_value":"0.696","numeric_value":"0.696","metric":"Spearman rho","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 1, mRNA-LM row, mRNA half-life Spearman column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558214+00:00","notes":"Resolved mRNA half-life column under the Spearman header spanning three tasks. mRNA-LM gives 0.696. Caption identifies average test performance across cross-validation splits; protein-expression AUROC is a different column.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 0.696.","artifact_sha256":"3a23de3c672ec162d13561c483f180a73b550d717256deffdc9099accec205fd","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11962594/fullTextXML"},"legacy_id":"b2-mrna-lm-2025","legacy_row":{"id":"b2-mrna-lm-2025","paper_id":"mrna-lm-2025","domain_id":"rna-transcriptomes","task":"mRNA half-life prediction","model":"mRNA-LM","model_version":"not stated in table","dataset":"mRNA half-life","dataset_version":"","split":"test set across CV splits","metric":"Spearman rho","value":"0.696","unit":"unitless","uncertainty":"","protocol":"average test performance across cross-validation splits","source_locator":"Table 1, mRNA-LM row, mRNA half-life Spearman column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11962594/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-mrnabert-2025","kind":"result","name":"mRNABERT · R-squared · human ultra-long mRNAs","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["translation-efficiency prediction"]},"source_ids":["mrnabert-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-mrnabert-2025"}],"attributes":{"printed_value":"0.669","numeric_value":"0.669","metric":"R-squared","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 2, mRNABERT (3066) row, Human R-squared column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558216+00:00","notes":"Resolved Human group and its R-squared subcolumn. mRNABERT (3066) reports 0.669; Human Spearman is 0.814 and Mouse R-squared is 0.649. Caption specifies ultra-long mRNA translation-efficiency prediction.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 0.669.","artifact_sha256":"ff08ba895b7080446c08a930548b48a0041ae990c222ebb07e6ba7dcaf48ad44","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12644827/fullTextXML"},"legacy_id":"b2-mrnabert-2025","legacy_row":{"id":"b2-mrnabert-2025","paper_id":"mrnabert-2025","domain_id":"rna-transcriptomes","task":"translation-efficiency prediction","model":"mRNABERT","model_version":"3066-nt input","dataset":"human ultra-long mRNAs","dataset_version":"","split":"paper evaluation","metric":"R-squared","value":"0.669","unit":"unitless","uncertainty":"","protocol":"human translation-efficiency regression at 3066-nt input","source_locator":"Table 2, mRNABERT (3066) row, Human R-squared column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12644827/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-mulan-2025","kind":"result","name":"MULAN-ESM2 S · AUC · HumanPPI","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["human protein-protein interaction prediction"]},"source_ids":["mulan-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-mulan-2025"}],"attributes":{"printed_value":"0.717","numeric_value":"0.717","metric":"AUC","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 2, MULAN-ESM2 S row, HumanPPI AUC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558217+00:00","notes":"Resolved the multirow header: HumanPPI uses AUC. MULAN-ESM2 S has 0.717; this is the small-model group, distinct from M and L variants.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 0.717.","artifact_sha256":"771a9a26ebda6f49ea266540e8dd6e6de0cbaef724de818ca6124a5f9c50d350","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12452268/fullTextXML"},"legacy_id":"b2-mulan-2025","legacy_row":{"id":"b2-mulan-2025","paper_id":"mulan-2025","domain_id":"proteins-complexes","task":"human protein-protein interaction prediction","model":"MULAN-ESM2 S","model_version":"small ESM2 backbone","dataset":"HumanPPI","dataset_version":"","split":"paper evaluation","metric":"AUC","value":"0.717","unit":"fraction","uncertainty":"","protocol":"MULAN sequence-structure model based on ESM2 8M","source_locator":"Table 2, MULAN-ESM2 S row, HumanPPI AUC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12452268/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-phylogpn-2025","kind":"result","name":"PhyloGPN · AUROC · ClinVar 3-prime UTR variants","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["ClinVar 3-prime UTR variant classification"]},"source_ids":["phylogpn-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-phylogpn-2025"}],"attributes":{"printed_value":"0.94","numeric_value":"0.94","metric":"AUROC","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 1, 3-prime UTR row, PhyloGPN AUROC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558218+00:00","notes":"Matched 3-prime UTR row and PhyloGPN column (0.94). Caption specifies log-likelihood-ratio predictions of ClinVar classes and explicitly defines each cell as AUROC.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 0.94.","artifact_sha256":"807f3a26cbfa9b5ce238d92164bd523302c67d1c5794b08273c51cca1acd4224","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11908359/fullTextXML"},"legacy_id":"b2-phylogpn-2025","legacy_row":{"id":"b2-phylogpn-2025","paper_id":"phylogpn-2025","domain_id":"dna-genomes","task":"ClinVar 3-prime UTR variant classification","model":"PhyloGPN","model_version":"not stated in table","dataset":"ClinVar 3-prime UTR variants","dataset_version":"","split":"paper evaluation","metric":"AUROC","value":"0.94","unit":"fraction","uncertainty":"","protocol":"log-likelihood-ratio scoring","source_locator":"Table 1, 3-prime UTR row, PhyloGPN AUROC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11908359/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-polya-glm-2025","kind":"result","name":"HyenaDNA · AUC · poly(A) Gene-Gene","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["polyadenylation site detection"]},"source_ids":["polya-glm-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-polya-glm-2025"}],"attributes":{"printed_value":"0.7510","numeric_value":"0.7510","metric":"AUC","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 1, Few-shot HyenaDNA row, G-G AUC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558220+00:00","notes":"Resolved Few-shot group, HyenaDNA row, and G-G subcolumn under AUC (0.7510). IG-G AUC is 0.7541. Caption states averages over five-fold cross-validation and distinguishes negative sampling regions.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 0.7510.","artifact_sha256":"e9ebd53d88837ad8d457881ffee918d2734dcae87d3c5cd03135947b6cf5dbde","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12799945/fullTextXML"},"legacy_id":"b2-polya-glm-2025","legacy_row":{"id":"b2-polya-glm-2025","paper_id":"polya-glm-2025","domain_id":"dna-genomes","task":"polyadenylation site detection","model":"HyenaDNA","model_version":"not stated in table","dataset":"poly(A) Gene-Gene","dataset_version":"","split":"5-fold cross-validation","metric":"AUC","value":"0.7510","unit":"fraction","uncertainty":"","protocol":"few-shot Gene-Gene negative-set comparison","source_locator":"Table 1, Few-shot HyenaDNA row, G-G AUC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12799945/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-rlsite-rna-binding-2025","kind":"result","name":"RLsite · AUC · T18","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA-small-molecule binding-site prediction"]},"source_ids":["rlsite-rna-binding-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-rlsite-rna-binding-2025"}],"attributes":{"printed_value":"0.828","numeric_value":"0.828","metric":"AUC","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 1, RLsite row, T18 AUC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558222+00:00","notes":"Matched RLsite and AUC (0.828). Caption explicitly identifies dataset T18; MCC 0.474 is a different metric.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 0.828.","artifact_sha256":"a50f344e253162ae43f51d7120cfb35a1d0f6114fd8176d760aceb6d05fd95bd","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12417085/fullTextXML"},"legacy_id":"b2-rlsite-rna-binding-2025","legacy_row":{"id":"b2-rlsite-rna-binding-2025","paper_id":"rlsite-rna-binding-2025","domain_id":"rna-transcriptomes","task":"RNA-small-molecule binding-site prediction","model":"RLsite","model_version":"not stated in table","dataset":"T18","dataset_version":"","split":"paper evaluation","metric":"AUC","value":"0.828","unit":"fraction","uncertainty":"","protocol":"RNA language-model plus graph-attention classifier","source_locator":"Table 1, RLsite row, T18 AUC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12417085/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-rnaret-2026","kind":"result","name":"RNAret · F1 · MirTarRAW","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["miRNA-mRNA interaction prediction"]},"source_ids":["rnaret-2026"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-rnaret-2026"}],"attributes":{"printed_value":"0.9622","numeric_value":"0.9622","metric":"F1","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 1, MirTarRAW / 5-mer RNAret row, F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558224+00:00","notes":"Resolved the MirTarRAW section, 5-mer RNAret row, and F1 column (0.9622), distinct from DeepMirTarLeft F1 0.9728. Methods confirm 72/8/20 train/validation/test partition for MirTarRAW.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 0.9622.","artifact_sha256":"e970e7322e07fb3c9d12efd315691cc5de5575a3f2616f4b788614c8c706dd0b","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13111708/fullTextXML"},"legacy_id":"b2-rnaret-2026","legacy_row":{"id":"b2-rnaret-2026","paper_id":"rnaret-2026","domain_id":"rna-transcriptomes","task":"miRNA-mRNA interaction prediction","model":"RNAret","model_version":"5-mer","dataset":"MirTarRAW","dataset_version":"","split":"held-out test","metric":"F1","value":"0.9622","unit":"fraction","uncertainty":"","protocol":"5-mer RNAret classifier; 72/8/20 train/validation/test split","source_locator":"Table 1, MirTarRAW / 5-mer RNAret row, F1 column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13111708/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-spin-protein-function-2026","kind":"result","name":"SPIN + ESM2-35M · F1 macro-weighted · TRX","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["protein function annotation"]},"source_ids":["spin-protein-function-2026"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-spin-protein-function-2026"}],"attributes":{"printed_value":"0.796","numeric_value":"0.796","metric":"F1 macro-weighted","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 1, ESM2-35M Test row, F1_m-w column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558225+00:00","notes":"Resolved Test group and macro-weighted F1 subcolumn (0.796) for frozen ESM2-35M in SPIN. Test weighted accuracy is 0.798. Methods define inverse-frequency class weighting for macro-weighted F1.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 0.796.","artifact_sha256":"9701843e93bf7fa3ead71e19693fb07d483f1022379871adfb04486783722a9d","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12970593/fullTextXML"},"legacy_id":"b2-spin-protein-function-2026","legacy_row":{"id":"b2-spin-protein-function-2026","paper_id":"spin-protein-function-2026","domain_id":"proteins-complexes","task":"protein function annotation","model":"SPIN + ESM2-35M","model_version":"ESM2-35M frozen","dataset":"TRX","dataset_version":"","split":"test set","metric":"F1 macro-weighted","value":"0.796","unit":"fraction","uncertainty":"","protocol":"frozen ESM2-35M backbone in SPIN","source_locator":"Table 1, ESM2-35M Test row, F1_m-w column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12970593/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-structure-informed-plm-2025","kind":"result","name":"structure-informed pLM · AUROC · variant-effects benchmark","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["protein variant-effect classification"]},"source_ids":["structure-informed-plm-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-structure-informed-plm-2025"}],"attributes":{"printed_value":".803","numeric_value":"0.803","metric":"AUROC","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"PMC12068927 HTML, Table4, AA+SS+RSA+CM row, AUROC column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source review, not human review","reviewed_at":"2026-09-16T10:45:41.099916+00:00","notes":"Full-text HTML succeeds although EuropePMC XMLreturned404. Row is mutation-site variables AA+SS+RSA+CM, not neighbouring environment variant. AUROC .803 is numerically equivalent to preserved legacy0.803. Source check, not experimental reproduction; do not claim original source printed leading zero.","evidence":"Table4 headers: Type, Variable(s), Spearman rho, AUROC, AUPRC. Parsed HTML row: AA+SS+RSA+CM | .552 | .803 | .792. Primary web rendering independently confirms columns.","artifact_sha256":"76082e1cd992d2c09c38f86d05aba575cc76c5022b53a297123b713bb1ce9267","retrieval_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12068927/"},"legacy_id":"b2-structure-informed-plm-2025","legacy_row":{"id":"b2-structure-informed-plm-2025","paper_id":"structure-informed-plm-2025","domain_id":"proteins-complexes","task":"protein variant-effect classification","model":"structure-informed pLM","model_version":"not stated in table","dataset":"variant-effects benchmark","dataset_version":"","split":"paper evaluation","metric":"AUROC","value":"0.803","unit":"fraction","uncertainty":"","protocol":"combined amino-acid, secondary structure, solvent accessibility and contact-map scoring","source_locator":"Table 4, AA+SS+RSA+CM row, AUROC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12068927/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"barcodebert-2026","kind":"source","name":"BarcodeBERT: transformers for biodiversity analyses","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["dna-genomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13008329/","version":"journal full text in PMC","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.1093/bioadv/vbag054","publication_status":"peer_reviewed","year":2026,"artifact_sha256":"493f9fe70b483780ba76d51ccf217d3ca83539c82b89917fd3ccebe2b6eb831d","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13008329/fullTextXML","artifact_retrieved_at":"2026-09-16T10:38:57.558051+00:00","legacy_paper":{"id":"barcodebert-2026","title":"BarcodeBERT: transformers for biodiversity analyses","year":2026,"publication_status":"peer_reviewed","version":"journal full text in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13008329/","primary_domain":"dna-genomes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.1093/bioadv/vbag054","notes":"Numeric result checked against Table 1 in primary full-text XML; journal/source: Bioinformatics Advances."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"birna-bert-2025","kind":"source","name":"BiRNA-BERT allows efficient RNA language modeling with adaptive tokenization","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["rna-transcriptomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12635123/","version":"journal full text in PMC","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.1038/s42003-025-08982-0","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"bf7dbc52b6515301c77010c513f13e676c38395ddc82c20310171f518690c152","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12635123/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:38.292Z","legacy_paper":{"id":"birna-bert-2025","title":"BiRNA-BERT allows efficient RNA language modeling with adaptive tokenization","year":2025,"publication_status":"peer_reviewed","version":"journal full text in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12635123/","primary_domain":"rna-transcriptomes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.1038/s42003-025-08982-0","notes":"Numeric result checked against Table 2 in primary full-text XML; journal/source: Communications Biology."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"boltz-stereochemistry-2025","kind":"source","name":"Improving Stereochemical Limitations in Protein–Ligand Complex Structure Prediction","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12658688/","version":"version of record","retrieved_at":"2026-09-15T23:25:00Z","doi":"10.1021/acsomega.5c07675","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"78a77b9a0ab8bfa371f5b9baef3f443f4590d6e71cf864d67e90e9ebdfa7fc1b","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12658688/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:16.555674+00:00","legacy_paper":{"id":"boltz-stereochemistry-2025","title":"Improving Stereochemical Limitations in Protein–Ligand Complex Structure Prediction","year":2025,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12658688/","primary_domain":"molecular-interactions","retrieved_utc":"2026-09-15T23:25:00Z","notes":"Primary full text via Europe PMC XML; venue: ACS Omega; PMC ID: PMC12658688.","doi":"10.1021/acsomega.5c07675"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"boltz1-2025","kind":"source","name":"Boltz-1 Democratizing Biomolecular Interaction Modeling","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11601547/","version":"PMC archival version PMC11601547.4","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.1101/2024.11.19.624167","publication_status":"preprint","year":2025,"artifact_sha256":"1ebf712314d9a1c678ded989cc95a0c00c0331e5ad8c9f63194bc9780971d214","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11601547/fullTextXML","artifact_retrieved_at":"2026-09-16T10:44:03.424864+00:00","legacy_paper":{"id":"boltz1-2025","title":"Boltz-1 Democratizing Biomolecular Interaction Modeling","year":2025,"publication_status":"preprint","version":"PMC archival version PMC11601547.4","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11601547/","primary_domain":"molecular-interactions","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: bioRxiv; PMC ID: PMC11601547.","doi":"10.1101/2024.11.19.624167"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"bpfold-2025","kind":"source","name":"Deep generalizable prediction of RNA secondary structure via base pair motif energy","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["rna-transcriptomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12216785/","version":"version of record","retrieved_at":"2026-09-15T23:25:00Z","doi":"10.1038/s41467-025-60048-1","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"976218bd172998a1a6e7ed1609ecb8cb2ee380fb48a8dc7b25bc05ea8b0a49af","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12216785/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:16.502000+00:00","legacy_paper":{"id":"bpfold-2025","title":"Deep generalizable prediction of RNA secondary structure via base pair motif energy","year":2025,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12216785/","primary_domain":"rna-transcriptomes","retrieved_utc":"2026-09-15T23:25:00Z","notes":"Primary full text via Europe PMC XML; venue: Nature Communications; PMC ID: PMC12216785.","doi":"10.1038/s41467-025-60048-1"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"cammiq-2022","kind":"source","name":"Strain level microbial detection and quantification with applications to single cell metagenomics","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC9616933/","version":"version of record","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.1038/s41467-022-33869-7","publication_status":"peer_reviewed","year":2022,"artifact_sha256":"f0939647ed3de995d58254f79472a612c21b0e1b2560a82783302aa1a148dde3","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC9616933/fullTextXML","artifact_retrieved_at":"2026-09-16T10:44:03.408237+00:00","legacy_paper":{"id":"cammiq-2022","title":"Strain level microbial detection and quantification with applications to single cell metagenomics","year":2022,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC9616933/","primary_domain":"microbes-communities","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: Nature Communications; PMC ID: PMC9616933.","doi":"10.1038/s41467-022-33869-7"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"catalog-baseline-kraken2","kind":"baseline","name":"Kraken2","description":"","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":["catalog-source-kraken2"],"links":[{"relation":"model","target_id":"catalog-model-kraken2"},{"relation":"applicable_to","target_id":"catalog-task-heldout-clade"},{"relation":"applicable_to","target_id":"catalog-task-phage-pathogen-reads"}],"attributes":{"baseline_type":"established_method","applicability":"proposed","requirements":"Public classifier; database build/version must be pinned separately.","missing_metadata":{"exact_protocol":"not_yet_extracted"}}} {"id":"catalog-baseline-scvi","kind":"baseline","name":"scVI","description":"","status":"discovered","facets":{"areas":["cells-tissues"]},"source_ids":["catalog-source-scvi"],"links":[{"relation":"model","target_id":"catalog-model-scvi"},{"relation":"applicable_to","target_id":"catalog-task-cell-reference-mapping"},{"relation":"applicable_to","target_id":"catalog-task-cell-batch-integration"}],"attributes":{"baseline_type":"established_method","applicability":"proposed","requirements":"Public software; train a task-specific model on the permitted split.","missing_metadata":{"exact_protocol":"not_yet_extracted"}}} {"id":"catalog-baseline-vina","kind":"baseline","name":"AutoDock Vina","description":"","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":["catalog-source-vina"],"links":[{"relation":"model","target_id":"catalog-model-vina"},{"relation":"applicable_to","target_id":"catalog-task-ligand-pose"}],"attributes":{"baseline_type":"established_method","applicability":"proposed","requirements":"Public docking software; receptor and ligand preparation required.","missing_metadata":{"exact_protocol":"not_yet_extracted"}}} {"id":"catalog-model-alphafold-3-server","kind":"model","name":"AlphaFold 3 Server","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["molecular-interactions"],"method_types":["foundation model"]},"source_ids":["catalog-source-alphafold-3-server"],"links":[],"attributes":{"entity_level":"family","version":"hosted server","reported_name":"AlphaFold 3 Server","access":"Manual, non-commercial server access; output terms restrict automated docking combinations.","method_type":"foundation model","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"AlphaFold 3 Server is a candidate method in the molecular-interactions catalogue. The linked resource identifies the project; a checkpoint-level profile has not yet been extracted.","sections":[{"title":"Available evidence","body":"The discovery record links to AlphaFold 3 Server official resource. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["catalog-source-alphafold-3-server"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"facts":[{"label":"Recorded configuration","value":"hosted server","source_ids":["catalog-source-alphafold-3-server"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["catalog-source-alphafold-3-server"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"coverage":"limited","gaps":["The linked official resource has not yielded a reviewed architecture or procedure for the configuration 'hosted server' in this release.","Unresolved registry fields: checkpoint revision (not yet extracted), training data (not yet extracted), licence (not yet extracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"catalog-model-alphagenome","kind":"model","name":"AlphaGenome","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["dna-genomes"],"method_types":["foundation model"]},"source_ids":["catalog-source-alphagenome"],"links":[],"attributes":{"entity_level":"family","version":"API / released weights","reported_name":"AlphaGenome","access":"Rate-limited, non-commercial API requires a key. Downloadable weights require accepting non-commercial model terms; local inference recommends an H100 GPU.","method_type":"foundation model","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"AlphaGenome is a candidate method in the dna-genomes catalogue. The linked resource identifies the project; a checkpoint-level profile has not yet been extracted.","sections":[{"title":"Available evidence","body":"The discovery record links to AlphaGenome official resource. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["catalog-source-alphagenome"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"facts":[{"label":"Recorded configuration","value":"API / released weights","source_ids":["catalog-source-alphagenome"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["catalog-source-alphagenome"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"coverage":"limited","gaps":["The linked official resource has not yielded a reviewed architecture or procedure for the configuration 'API / released weights' in this release.","Unresolved registry fields: checkpoint revision (not yet extracted), training data (not yet extracted), licence (not yet extracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"catalog-model-boltz-2","kind":"model","name":"Boltz-2","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["molecular-interactions"],"method_types":["foundation model"]},"source_ids":["catalog-source-boltz-2"],"links":[{"relation":"variant_of","target_id":"discovery-model-boltz"}],"attributes":{"entity_level":"family","version":"released weights","reported_name":"Boltz-2","access":"Public MIT code and weights; substantial compute required.","method_type":"foundation model","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"Boltz is a biomolecular interaction model family. Boltz-2 adds affinity prediction to complex-structure prediction.","sections":[{"title":"Boltz-2 architecture","body":"The Boltz-2 implementation combines molecular and alignment features with a Pairformer module. A conditioned diffusion module predicts coordinates; a separate affinity module produces binding outputs. This architecture description applies to Boltz-2, not automatically to every Boltz family release.","source_ids":["src-discovery-jwohlwend-boltz"],"source_locator":"src/boltz/model/models/boltz2.py at the pinned repository revision: MSAModule, PairformerModule, DiffusionConditioning and AffinityModule; README Inference"},{"title":"How it works","body":"The documented YAML input describes the biomolecules and requested properties. Structure prediction and affinity outputs are distinct: one affinity output estimates binding strength, while another classifies binders against decoys.","source_ids":["src-discovery-jwohlwend-boltz"],"source_locator":"README.md: Introduction, Inference, Understanding the affinity prediction and License"}],"facts":[{"label":"Access","value":"Repository states code and models use the MIT licence","source_ids":["src-discovery-jwohlwend-boltz"],"source_locator":"README.md: Introduction, Inference, Understanding the affinity prediction and License"},{"label":"Configuration in this record","value":"released weights","source_ids":["src-discovery-jwohlwend-boltz"],"source_locator":"README.md: Introduction, Inference, Understanding the affinity prediction and License"}],"strengths":[{"text":"Supports structure and affinity workflows in one openly distributed project.","source_ids":["src-discovery-jwohlwend-boltz"],"source_locator":"README.md: Introduction, Inference, Understanding the affinity prediction and License"}],"limitations":[{"text":"Binder probability and affinity regression are trained with different supervision and must not be compared as the same metric. Unqualified CLI calls select the latest model.","source_ids":["src-discovery-jwohlwend-boltz"],"source_locator":"README.md: Introduction, Inference, Understanding the affinity prediction and License"}],"diagram":{"title":"Conceptual procedure","steps":["Molecular inputs","MSA / Pairformer features","Coordinate diffusion","Structure","Separate affinity module"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-jwohlwend-boltz"],"source_locator":"Pinned repository src/boltz/model/models/boltz2.py, module construction and forward; README affinity prediction"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}}} {"id":"catalog-model-chai-1","kind":"model","name":"Chai-1","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["proteins-complexes","molecular-interactions"],"method_types":["foundation model"]},"source_ids":["catalog-source-chai-1"],"links":[],"attributes":{"entity_level":"family","version":"released weights","reported_name":"Chai-1","access":"Public code and weights under Apache 2.0; substantial compute required.","method_type":"foundation model","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"Chai-1 is a candidate method in the proteins-complexes, molecular-interactions catalogue. The linked resource identifies the project; a checkpoint-level profile has not yet been extracted.","sections":[{"title":"Available evidence","body":"The discovery record links to Chai-1 official resource. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["catalog-source-chai-1"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"facts":[{"label":"Recorded configuration","value":"released weights","source_ids":["catalog-source-chai-1"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["catalog-source-chai-1"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"coverage":"limited","gaps":["The linked official resource has not yielded a reviewed architecture or procedure for the configuration 'released weights' in this release.","Unresolved registry fields: checkpoint revision (not yet extracted), training data (not yet extracted), licence (not yet extracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"catalog-model-diffdock-l","kind":"model","name":"DiffDock-L","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["molecular-interactions"],"method_types":["specialist"]},"source_ids":["catalog-source-diffdock-l"],"links":[],"attributes":{"entity_level":"family","version":"2024 release","reported_name":"DiffDock-L","access":"Public pose-prediction code and weights; no native affinity prediction.","method_type":"specialist","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"DiffDock-L is a candidate method in the molecular-interactions catalogue. The linked resource identifies the project; a checkpoint-level profile has not yet been extracted.","sections":[{"title":"Available evidence","body":"The discovery record links to DiffDock-L official resource. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["catalog-source-diffdock-l"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"facts":[{"label":"Recorded configuration","value":"2024 release","source_ids":["catalog-source-diffdock-l"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["catalog-source-diffdock-l"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"coverage":"limited","gaps":["The linked official resource has not yielded a reviewed architecture or procedure for the configuration '2024 release' in this release.","Unresolved registry fields: checkpoint revision (not yet extracted), training data (not yet extracted), licence (not yet extracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"catalog-model-dnabert-2","kind":"model","name":"DNABERT-2","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["dna-genomes"],"method_types":["foundation model"]},"source_ids":["catalog-source-dnabert-2"],"links":[{"relation":"variant_of","target_id":"discovery-model-dnabert-2"}],"attributes":{"entity_level":"family","version":"117M","reported_name":"DNABERT-2","access":"Public checkpoint; remote model code needs review before local use.","method_type":"foundation model","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"DNABERT-2 is a DNA encoder pretrained on sequences from multiple species. It supplies representations that can be adapted to genomic tasks.","sections":[{"title":"How it works","body":"DNA is compressed into variable-length byte-pair tokens. A BERT-style encoder uses ALiBi positional biases; masked-language pretraining learns contextual features. Sequence pooling or a separately trained prediction head produces task outputs.","source_ids":["src-discovery-magics-lab-dnabert-2"],"source_locator":"README.md: Introduction, Model and Data, Quick Start, Pre-Training and Fine-tune"}],"facts":[{"label":"Released model","value":"DNABERT-2-117M","source_ids":["src-discovery-magics-lab-dnabert-2"],"source_locator":"README.md: Introduction, Model and Data, Quick Start, Pre-Training and Fine-tune"},{"label":"Objective","value":"Masked-language pretraining","source_ids":["src-discovery-magics-lab-dnabert-2"],"source_locator":"README.md: Introduction, Model and Data, Quick Start, Pre-Training and Fine-tune"},{"label":"Configuration in this record","value":"117M","source_ids":["src-discovery-magics-lab-dnabert-2"],"source_locator":"README.md: Introduction, Model and Data, Quick Start, Pre-Training and Fine-tune"}],"strengths":[{"text":"One released encoder can support embedding extraction and supervised adaptation across several genomic tasks.","source_ids":["src-discovery-magics-lab-dnabert-2"],"source_locator":"README.md: Introduction, Model and Data, Quick Start, Pre-Training and Fine-tune"}],"limitations":[{"text":"An encoder embedding is not a splice-impact prediction. Pooling, sequence context and the supervised head are part of the evaluated pipeline.","source_ids":["src-discovery-magics-lab-dnabert-2"],"source_locator":"README.md: Introduction, Model and Data, Quick Start, Pre-Training and Fine-tune"}],"diagram":{"title":"Conceptual procedure","steps":["DNA sequence","Byte-pair tokens","BERT encoder with ALiBi","Token or pooled embeddings","Task-specific head"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-magics-lab-dnabert-2"],"source_locator":"README.md: Introduction, Model and Data, Quick Start, Pre-Training and Fine-tune"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}}} {"id":"catalog-model-esm-2","kind":"model","name":"ESM-2","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["proteins-complexes"],"method_types":["foundation model"]},"source_ids":["catalog-source-esm-2"],"links":[{"relation":"variant_of","target_id":"discovery-model-esm-2"}],"attributes":{"entity_level":"family","version":"8M","reported_name":"ESM-2","access":"Public checkpoint; small 8M variant suits a local pilot.","method_type":"foundation model","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"ESM-2 is a family of protein sequence transformers that produce residue-level and sequence-level representations.","sections":[{"title":"How it works","body":"Amino-acid tokens pass through a pretrained transformer. Hidden states can be retained for each residue or pooled for a whole protein; downstream tasks need an explicit scoring rule or predictor. ESMFold adds a structure-prediction system and is a separate pipeline.","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md: Main models, Getting started, Compute embeddings and Pre-trained Models"}],"facts":[{"label":"Training resource","value":"UniRef-derived protein sequences","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md: Main models, Getting started, Compute embeddings and Pre-trained Models"},{"label":"Configuration distinction","value":"The catalogue 8M entry is not the 650M or 15B checkpoint","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md: Main models, Getting started, Compute embeddings and Pre-trained Models"},{"label":"Configuration in this record","value":"8M","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md: Main models, Getting started, Compute embeddings and Pre-trained Models"}],"strengths":[{"text":"Embeddings can be extracted directly from individual sequences; the repository provides several model sizes.","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md: Main models, Getting started, Compute embeddings and Pre-trained Models"}],"limitations":[{"text":"Different parameter sizes, pooling methods and supervised heads are not interchangeable evaluations.","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md: Main models, Getting started, Compute embeddings and Pre-trained Models"}],"diagram":{"title":"Conceptual procedure","steps":["Protein sequence","Amino-acid tokens","ESM-2 transformer","Residue embeddings","Pooling or task predictor"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md: Main models, Getting started, Compute embeddings and Pre-trained Models"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}}} {"id":"catalog-model-esmfold","kind":"model","name":"ESMFold","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["proteins-complexes"],"method_types":["foundation model"]},"source_ids":["catalog-source-esmfold"],"links":[{"relation":"variant_of","target_id":"discovery-model-esmfold"}],"attributes":{"entity_level":"family","version":"v1","reported_name":"ESMFold","access":"Public checkpoint; materially larger than ESM-2 8M.","method_type":"foundation model","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"ESMFold predicts protein structures from individual amino-acid sequences using an ESM-2 representation model and a folding system.","sections":[{"title":"How it works","body":"The sequence is embedded and converted into a three-dimensional structure. The implementation exposes recycling and chunking controls; those choices affect memory use and the exact evaluated run.","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md: ESMFold Structure Prediction and Main models"}],"facts":[{"label":"Primary output","value":"PDB structure with confidence information","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md: ESMFold Structure Prediction and Main models"},{"label":"Configuration in this record","value":"v1","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md: ESMFold Structure Prediction and Main models"}],"strengths":[{"text":"The documented inference interface produces a structure directly from sequence.","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md: ESMFold Structure Prediction and Main models"}],"limitations":[{"text":"Long sequences and larger batches can exceed device memory. Version v0 and v1 refer to different released models.","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md: ESMFold Structure Prediction and Main models"}],"diagram":{"title":"Conceptual procedure","steps":["Protein sequence","ESM-2 features","Folding system","Recycling","Predicted structure"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md: ESMFold Structure Prediction and Main models"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}}} {"id":"catalog-model-evo-2","kind":"model","name":"Evo 2","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["dna-genomes","microbes-communities"],"method_types":["foundation model"]},"source_ids":["catalog-source-evo-2"],"links":[],"attributes":{"entity_level":"family","version":"7B","reported_name":"Evo 2","access":"Public checkpoints; official local inference needs CUDA hardware and substantial memory.","method_type":"foundation model","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"Evo 2 models and generates DNA at nucleotide resolution using the StripedHyena 2 architecture.","sections":[{"title":"How it works","body":"An autoregressive sequence model predicts successive nucleotides from preceding context. Scoring and generation use the selected released checkpoint; context length and device requirements depend on that configuration.","source_ids":["src-discovery-arcinstitute-evo2"],"source_locator":"README.md: introduction, Models and Installation"}],"facts":[{"label":"Training resource","value":"OpenGenome2","source_ids":["src-discovery-arcinstitute-evo2"],"source_locator":"README.md: introduction, Models and Installation"},{"label":"Objective","value":"Autoregressive sequence prediction","source_ids":["src-discovery-arcinstitute-evo2"],"source_locator":"README.md: introduction, Models and Installation"},{"label":"Configuration in this record","value":"7B","source_ids":["src-discovery-arcinstitute-evo2"],"source_locator":"README.md: introduction, Models and Installation"}],"strengths":[{"text":"The family is designed for long-context sequence modelling, with released inference code.","source_ids":["src-discovery-arcinstitute-evo2"],"source_locator":"README.md: introduction, Models and Installation"}],"limitations":[{"text":"A family-level maximum context length does not establish the settings used by a particular published evaluation.","source_ids":["src-discovery-arcinstitute-evo2"],"source_locator":"README.md: introduction, Models and Installation"}],"diagram":{"title":"Conceptual procedure","steps":["DNA nucleotides","StripedHyena 2","Autoregressive predictions","Sequence scoring or generation"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-arcinstitute-evo2"],"source_locator":"README.md: introduction, Models and Installation"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}}} {"id":"catalog-model-gears","kind":"model","name":"GEARS","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["cells-tissues"],"method_types":["specialist"]},"source_ids":["catalog-source-gears"],"links":[{"relation":"family","target_id":"discovery-model-gears"}],"attributes":{"entity_level":"family","version":"published implementation","reported_name":"GEARS","access":"Public code; task-specific training data required.","method_type":"specialist","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"GEARS predicts transcriptional responses to genetic perturbations using single-cell perturbation-screen data.","sections":[{"title":"How it works","body":"A task-specific model is trained on measured perturbations, then predicts gene-expression responses for requested single or combined perturbations. Training composition determines what generalisation question is being tested.","source_ids":["src-discovery-snap-stanford-gears"],"source_locator":"README.md: introduction and Important notes"}],"facts":[{"label":"Required evidence","value":"Perturbation identities and cells per condition","source_ids":["src-discovery-snap-stanford-gears"],"source_locator":"README.md: introduction and Important notes"},{"label":"Configuration in this record","value":"published implementation","source_ids":["src-discovery-snap-stanford-gears"],"source_locator":"README.md: introduction and Important notes"}],"strengths":[{"text":"The implementation explicitly supports single-gene and multi-gene perturbation workflows.","source_ids":["src-discovery-snap-stanford-gears"],"source_locator":"README.md: introduction and Important notes"}],"limitations":[{"text":"The maintainers state that cross-cell-type transfer is unsupported and that reliable combinatorial prediction needs some combinatorial training data.","source_ids":["src-discovery-snap-stanford-gears"],"source_locator":"README.md: introduction and Important notes"}],"diagram":{"title":"Conceptual procedure","steps":["Perturbation-screen cells","Training perturbations","GEARS predictor","Requested perturbation","Expression response"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-snap-stanford-gears"],"source_locator":"README.md: introduction and Important notes"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}}} {"id":"catalog-model-geneformer","kind":"model","name":"Geneformer","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["cells-tissues"],"method_types":["foundation model"]},"source_ids":["catalog-source-geneformer"],"links":[],"attributes":{"entity_level":"family","version":"published checkpoints","reported_name":"Geneformer","access":"Public checkpoints; specify exact version before evaluation.","method_type":"foundation model","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"Geneformer is a candidate method in the cells-tissues catalogue. The linked resource identifies the project; a checkpoint-level profile has not yet been extracted.","sections":[{"title":"Available evidence","body":"The discovery record links to Geneformer official resource. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["catalog-source-geneformer"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"facts":[{"label":"Recorded configuration","value":"published checkpoints","source_ids":["catalog-source-geneformer"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["catalog-source-geneformer"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"coverage":"limited","gaps":["The linked official resource has not yielded a reviewed architecture or procedure for the configuration 'published checkpoints' in this release.","Unresolved registry fields: checkpoint revision (not yet extracted), training data (not yet extracted), licence (not yet extracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"catalog-model-kraken2","kind":"model","name":"Kraken2","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["microbes-communities"],"method_types":["baseline"]},"source_ids":["catalog-source-kraken2"],"links":[],"attributes":{"entity_level":"family","version":"current database pinned at run time","reported_name":"Kraken2","access":"Public classifier; database build/version must be pinned separately.","method_type":"baseline","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"Kraken2 is a candidate method in the microbes-communities catalogue. The linked resource identifies the project; a checkpoint-level profile has not yet been extracted.","sections":[{"title":"Available evidence","body":"The discovery record links to Kraken2 official resource. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["catalog-source-kraken2"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"facts":[{"label":"Recorded configuration","value":"current database pinned at run time","source_ids":["catalog-source-kraken2"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["catalog-source-kraken2"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"coverage":"limited","gaps":["The linked official resource has not yielded a reviewed architecture or procedure for the configuration 'current database pinned at run time' in this release.","Unresolved registry fields: checkpoint revision (not yet extracted), training data (not yet extracted), licence (not yet extracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"catalog-model-metagene-1","kind":"model","name":"METAGENE-1","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["microbes-communities"],"method_types":["foundation model"]},"source_ids":["catalog-source-metagene-1"],"links":[],"attributes":{"entity_level":"family","version":"6B","reported_name":"METAGENE-1","access":"Public Apache 2.0 checkpoint; 512-token context and large local memory requirement.","method_type":"foundation model","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"METAGENE-1 is a candidate method in the microbes-communities catalogue. The linked resource identifies the project; a checkpoint-level profile has not yet been extracted.","sections":[{"title":"Available evidence","body":"The discovery record links to METAGENE-1 official resource. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["catalog-source-metagene-1"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"facts":[{"label":"Recorded configuration","value":"6B","source_ids":["catalog-source-metagene-1"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["catalog-source-metagene-1"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"coverage":"limited","gaps":["The linked official resource has not yielded a reviewed architecture or procedure for the configuration '6B' in this release.","Unresolved registry fields: checkpoint revision (not yet extracted), training data (not yet extracted), licence (not yet extracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"catalog-model-metaphlan","kind":"model","name":"MetaPhlAn","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["microbes-communities"],"method_types":["specialist"]},"source_ids":["catalog-source-metaphlan"],"links":[],"attributes":{"entity_level":"family","version":"current marker database pinned at run time","reported_name":"MetaPhlAn","access":"Public profiler; marker database version must be pinned separately.","method_type":"specialist","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"MetaPhlAn is a candidate method in the microbes-communities catalogue. The linked resource identifies the project; a checkpoint-level profile has not yet been extracted.","sections":[{"title":"Available evidence","body":"The discovery record links to MetaPhlAn official resource. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["catalog-source-metaphlan"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"facts":[{"label":"Recorded configuration","value":"current marker database pinned at run time","source_ids":["catalog-source-metaphlan"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["catalog-source-metaphlan"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"coverage":"limited","gaps":["The linked official resource has not yielded a reviewed architecture or procedure for the configuration 'current marker database pinned at run time' in this release.","Unresolved registry fields: checkpoint revision (not yet extracted), training data (not yet extracted), licence (not yet extracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"catalog-model-mimic","kind":"model","name":"MIMIC","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["rna-transcriptomes","proteins-complexes"],"method_types":["foundation model"]},"source_ids":["catalog-source-mimic"],"links":[],"attributes":{"entity_level":"family","version":"1.0","reported_name":"MIMIC","access":"Public MIT code and 1.25B-parameter weights; large local memory requirement.","method_type":"foundation model","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"MIMIC is a candidate method in the rna-transcriptomes, proteins-complexes catalogue. The linked resource identifies the project; a checkpoint-level profile has not yet been extracted.","sections":[{"title":"Available evidence","body":"The discovery record links to MIMIC official resource. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["catalog-source-mimic"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"facts":[{"label":"Recorded configuration","value":"1.0","source_ids":["catalog-source-mimic"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["catalog-source-mimic"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"coverage":"limited","gaps":["The linked official resource has not yielded a reviewed architecture or procedure for the configuration '1.0' in this release.","Unresolved registry fields: checkpoint revision (not yet extracted), training data (not yet extracted), licence (not yet extracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"catalog-model-mrna-fm","kind":"model","name":"mRNA-FM","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["rna-transcriptomes"],"method_types":["foundation model"]},"source_ids":["catalog-source-mrna-fm"],"links":[{"relation":"variant_of","target_id":"discovery-model-rna-fm"}],"attributes":{"entity_level":"family","version":"codon-tokenised","reported_name":"mRNA-FM","access":"Public checkpoint trained on coding sequences (CDS); input must be codon aligned. UTR-only sequences are outside its training modality.","method_type":"foundation model","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"mRNA-FM is the coding-sequence extension of RNA-FM, intended to represent messenger RNA coding regions.","sections":[{"title":"How it works","body":"Coding sequences are converted into model tokens and processed by the pretrained sequence encoder. Its output supplies embeddings for a downstream predictor. The input preparation differs from the ncRNA model.","source_ids":["src-discovery-ml4bio-rna-fm"],"source_locator":"README.md: introduction, mRNA-FM subsection"}],"facts":[{"label":"Training modality","value":"Repository reports 45 million mRNA coding sequences","source_ids":["src-discovery-ml4bio-rna-fm"],"source_locator":"README.md: introduction, mRNA-FM subsection"},{"label":"Configuration in this record","value":"codon-tokenised","source_ids":["src-discovery-ml4bio-rna-fm"],"source_locator":"README.md: introduction, mRNA-FM subsection"}],"strengths":[{"text":"Provides a representation specifically pretrained on coding RNA.","source_ids":["src-discovery-ml4bio-rna-fm"],"source_locator":"README.md: introduction, mRNA-FM subsection"}],"limitations":[{"text":"Coding-sequence training does not establish performance on UTR-only inputs or other non-coding RNA.","source_ids":["src-discovery-ml4bio-rna-fm"],"source_locator":"README.md: introduction, mRNA-FM subsection"}],"diagram":{"title":"Conceptual procedure","steps":["Coding sequence","mRNA-FM input tokens","Pretrained encoder","Embeddings","Downstream predictor"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-ml4bio-rna-fm"],"source_locator":"README.md: introduction, mRNA-FM subsection"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}}} {"id":"catalog-model-nt-v2","kind":"model","name":"Nucleotide Transformer v2","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["dna-genomes"],"method_types":["foundation model"]},"source_ids":["catalog-source-nt-v2"],"links":[],"attributes":{"entity_level":"family","version":"50M multi-species","reported_name":"Nucleotide Transformer v2","access":"Public checkpoint.","method_type":"foundation model","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"Nucleotide Transformer v2 is a candidate method in the dna-genomes catalogue. The linked resource identifies the project; a checkpoint-level profile has not yet been extracted.","sections":[{"title":"Available evidence","body":"The discovery record links to Nucleotide Transformer v2 official resource. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["catalog-source-nt-v2"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"facts":[{"label":"Recorded configuration","value":"50M multi-species","source_ids":["catalog-source-nt-v2"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["catalog-source-nt-v2"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"coverage":"limited","gaps":["The linked official resource has not yielded a reviewed architecture or procedure for the configuration '50M multi-species' in this release.","Unresolved registry fields: checkpoint revision (not yet extracted), training data (not yet extracted), licence (not yet extracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"catalog-model-pangolin","kind":"model","name":"Pangolin","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["dna-genomes"],"method_types":["specialist"]},"source_ids":["catalog-source-pangolin"],"links":[{"relation":"family","target_id":"discovery-model-pangolin"}],"attributes":{"entity_level":"family","version":"published checkpoints","reported_name":"Pangolin","access":"Public specialist code and models under GPL-3.0.","method_type":"specialist","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"Pangolin predicts changes in splice-site strength from DNA variants. It accepts variant files or custom sequence inputs.","sections":[{"title":"Architecture","body":"Pangolin uses 16 residual blocks with dilated convolutions and skip connections. Separate outputs estimate splice-site probability and usage across heart, liver, brain and testis. The published model was trained using sequence and splicing measurements from human, rhesus macaque, rat and mouse.","source_ids":["src-discovery-tkzeng-pangolin"],"source_locator":"Original paper linked in README: https://doi.org/10.1186/s13059-022-02664-4, Methods: Deep neural network architecture; Results: Pangolin predicts splice site usage"},{"title":"How it works","body":"Reference genome and transcript annotation define the sequence context. The neural predictor estimates splice-site strength; the command-line tool reports the largest positive and negative changes near each variant. Masking optionally removes particular gains and losses at annotated sites.","source_ids":["src-discovery-tkzeng-pangolin"],"source_locator":"README.md: introduction and Usage (command-line), including supported variants and --mask"}],"facts":[{"label":"Masking","value":"Default mask=True; a mask=False evaluation is a distinct configuration","source_ids":["src-discovery-tkzeng-pangolin"],"source_locator":"README.md: introduction and Usage (command-line), including supported variants and --mask"},{"label":"Configuration in this record","value":"published checkpoints","source_ids":["src-discovery-tkzeng-pangolin"],"source_locator":"README.md: introduction and Usage (command-line), including supported variants and --mask"}],"strengths":[{"text":"Provides changes in splice strength and their positions, with configurable scoring distance and masking.","source_ids":["src-discovery-tkzeng-pangolin"],"source_locator":"README.md: introduction and Usage (command-line), including supported variants and --mask"}],"limitations":[{"text":"Only substitutions and simple indels are supported by the documented interface. Missing gene annotations, reference mismatches and chromosome-edge cases can exclude variants.","source_ids":["src-discovery-tkzeng-pangolin"],"source_locator":"README.md: introduction and Usage (command-line), including supported variants and --mask"}],"diagram":{"title":"Conceptual procedure","steps":["DNA context","Dilated residual convolutions","Tissue-specific outputs","Splice strength","Variant-induced change"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-tkzeng-pangolin"],"source_locator":"Original paper linked in README: https://doi.org/10.1186/s13059-022-02664-4, Figure 1 and Methods: Deep neural network architecture; README Usage"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}}} {"id":"catalog-model-prokbert","kind":"model","name":"ProkBERT","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["microbes-communities"],"method_types":["foundation model"]},"source_ids":["catalog-source-prokbert"],"links":[],"attributes":{"entity_level":"family","version":"mini","reported_name":"ProkBERT","access":"Public model family and mini checkpoint.","method_type":"foundation model","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"ProkBERT is a candidate method in the microbes-communities catalogue. The linked resource identifies the project; a checkpoint-level profile has not yet been extracted.","sections":[{"title":"Available evidence","body":"The discovery record links to ProkBERT official resource. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["catalog-source-prokbert"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"facts":[{"label":"Recorded configuration","value":"mini","source_ids":["catalog-source-prokbert"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["catalog-source-prokbert"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"coverage":"limited","gaps":["The linked official resource has not yielded a reviewed architecture or procedure for the configuration 'mini' in this release.","Unresolved registry fields: checkpoint revision (not yet extracted), training data (not yet extracted), licence (not yet extracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"catalog-model-proteinmpnn","kind":"model","name":"ProteinMPNN","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["proteins-complexes"],"method_types":["specialist"]},"source_ids":["catalog-source-proteinmpnn"],"links":[{"relation":"variant_of","target_id":"discovery-model-proteinmpnn"}],"attributes":{"entity_level":"family","version":"v_48_020","reported_name":"ProteinMPNN","access":"Public code and checkpoints; requires a suitable protein structure.","method_type":"specialist","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"ProteinMPNN designs amino-acid sequences for a supplied protein backbone, with controls for fixed residues and chains.","sections":[{"title":"How it works","body":"A parsed structure and design constraints are supplied to the sequence-design model. It samples amino-acid sequences conditional on the backbone; sampling temperature changes diversity. Full-backbone and Cα-only weights are separate configurations.","source_ids":["src-discovery-dauparas-proteinmpnn"],"source_locator":"README.md: Full protein backbone models, CA only models and protein_mpnn_run.py arguments"}],"facts":[{"label":"Catalogue weight name","value":"v_48_020","source_ids":["src-discovery-dauparas-proteinmpnn"],"source_locator":"README.md: Full protein backbone models, CA only models and protein_mpnn_run.py arguments"},{"label":"Output","value":"Designed sequences and model scores","source_ids":["src-discovery-dauparas-proteinmpnn"],"source_locator":"README.md: Full protein backbone models, CA only models and protein_mpnn_run.py arguments"},{"label":"Configuration in this record","value":"v_48_020","source_ids":["src-discovery-dauparas-proteinmpnn"],"source_locator":"README.md: Full protein backbone models, CA only models and protein_mpnn_run.py arguments"}],"strengths":[{"text":"Allows selected chains and positions to be redesigned while retaining specified residues.","source_ids":["src-discovery-dauparas-proteinmpnn"],"source_locator":"README.md: Full protein backbone models, CA only models and protein_mpnn_run.py arguments"}],"limitations":[{"text":"Requires a suitable input structure. Sequence generation does not itself demonstrate folding, activity or experimental success.","source_ids":["src-discovery-dauparas-proteinmpnn"],"source_locator":"README.md: Full protein backbone models, CA only models and protein_mpnn_run.py arguments"}],"diagram":{"title":"Conceptual procedure","steps":["Backbone structure","Chain / residue constraints","ProteinMPNN","Conditional sequence sampling","Designed sequences"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-dauparas-proteinmpnn"],"source_locator":"README.md: Full protein backbone models, CA only models and protein_mpnn_run.py arguments"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}}} {"id":"catalog-model-rhofold","kind":"model","name":"RhoFold+","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["rna-transcriptomes"],"method_types":["specialist"]},"source_ids":["catalog-source-rhofold"],"links":[],"attributes":{"entity_level":"family","version":"pretrained","reported_name":"RhoFold+","access":"Public code and checkpoint instructions.","method_type":"specialist","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"RhoFold+ is a candidate method in the rna-transcriptomes catalogue. The linked resource identifies the project; a checkpoint-level profile has not yet been extracted.","sections":[{"title":"Available evidence","body":"The discovery record links to RhoFold+ official resource. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["catalog-source-rhofold"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"facts":[{"label":"Recorded configuration","value":"pretrained","source_ids":["catalog-source-rhofold"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["catalog-source-rhofold"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"coverage":"limited","gaps":["The linked official resource has not yielded a reviewed architecture or procedure for the configuration 'pretrained' in this release.","Unresolved registry fields: checkpoint revision (not yet extracted), training data (not yet extracted), licence (not yet extracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"catalog-model-rna-fm","kind":"model","name":"RNA-FM","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["rna-transcriptomes"],"method_types":["foundation model"]},"source_ids":["catalog-source-rna-fm"],"links":[{"relation":"family","target_id":"discovery-model-rna-fm"}],"attributes":{"entity_level":"family","version":"ncRNA","reported_name":"RNA-FM","access":"Public code and checkpoint instructions.","method_type":"foundation model","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"RNA-FM is a pretrained RNA sequence encoder for structural and functional representation learning.","sections":[{"title":"How it works","body":"A BERT-style transformer encodes RNA tokens into contextual embeddings after self-supervised sequence training. Structural or functional predictions require the corresponding downstream model; RNA-FM alone should not be labelled as a complete 3D folding pipeline.","source_ids":["src-discovery-ml4bio-rna-fm"],"source_locator":"README.md: introduction and Foundation Models and Extended Ecosystem"}],"facts":[{"label":"RNA-FM training","value":"Repository reports more than 23 million non-coding RNA sequences","source_ids":["src-discovery-ml4bio-rna-fm"],"source_locator":"README.md: introduction and Foundation Models and Extended Ecosystem"},{"label":"Configuration in this record","value":"ncRNA","source_ids":["src-discovery-ml4bio-rna-fm"],"source_locator":"README.md: introduction and Foundation Models and Extended Ecosystem"}],"strengths":[{"text":"Reusable representations do not require experimental labels during pretraining.","source_ids":["src-discovery-ml4bio-rna-fm"],"source_locator":"README.md: introduction and Foundation Models and Extended Ecosystem"}],"limitations":[{"text":"The ncRNA encoder and the coding-sequence mRNA-FM extension have different training modalities and should not share checkpoint identities.","source_ids":["src-discovery-ml4bio-rna-fm"],"source_locator":"README.md: introduction and Foundation Models and Extended Ecosystem"}],"diagram":{"title":"Conceptual procedure","steps":["RNA sequence","RNA tokens","Pretrained transformer","Contextual embeddings","Task-specific predictor"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-ml4bio-rna-fm"],"source_locator":"README.md: introduction and Foundation Models and Extended Ecosystem"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}}} {"id":"catalog-model-scfoundation","kind":"model","name":"scFoundation","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["cells-tissues"],"method_types":["foundation model"]},"source_ids":["catalog-source-scfoundation"],"links":[],"attributes":{"entity_level":"family","version":"100M","reported_name":"scFoundation","access":"Public code; model weights have separate terms that must be checked.","method_type":"foundation model","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"scFoundation is a candidate method in the cells-tissues catalogue. The linked resource identifies the project; a checkpoint-level profile has not yet been extracted.","sections":[{"title":"Available evidence","body":"The discovery record links to scFoundation official resource. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["catalog-source-scfoundation"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"facts":[{"label":"Recorded configuration","value":"100M","source_ids":["catalog-source-scfoundation"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["catalog-source-scfoundation"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"coverage":"limited","gaps":["The linked official resource has not yielded a reviewed architecture or procedure for the configuration '100M' in this release.","Unresolved registry fields: checkpoint revision (not yet extracted), training data (not yet extracted), licence (not yet extracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"catalog-model-scgpt","kind":"model","name":"scGPT","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["cells-tissues"],"method_types":["foundation model"]},"source_ids":["catalog-source-scgpt"],"links":[{"relation":"variant_of","target_id":"discovery-model-scgpt"}],"attributes":{"entity_level":"family","version":"whole-human","reported_name":"scGPT","access":"Public code and downloadable checkpoints; use the unfine-tuned whole-human model for a new task.","method_type":"foundation model","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"scGPT learns representations of genes and cells from single-cell measurements. Its pretrained checkpoints support task-specific adaptation.","sections":[{"title":"How it works","body":"Gene identifiers and expression values are encoded together and processed by a transformer. The resulting representations support cell embeddings or task heads. Vocabulary, preprocessing and the selected checkpoint must accompany any result.","source_ids":["src-discovery-bowang-lab-scgpt"],"source_locator":"README.md: Pretrained scGPT checkpoints and Tutorials; scgpt/model/model.py at cebd6fae655b9c585a4807daa3ac31bb764f06b4, lines 79–188"}],"facts":[{"label":"Whole-human training","value":"Repository reports 33 million normal human cells","source_ids":["src-discovery-bowang-lab-scgpt"],"source_locator":"README.md: Pretrained scGPT checkpoints and Tutorials; scgpt/model/model.py at cebd6fae655b9c585a4807daa3ac31bb764f06b4, lines 79–188"},{"label":"Configuration in this record","value":"whole-human","source_ids":["src-discovery-bowang-lab-scgpt"],"source_locator":"README.md: Pretrained scGPT checkpoints and Tutorials; scgpt/model/model.py at cebd6fae655b9c585a4807daa3ac31bb764f06b4, lines 79–188"}],"strengths":[{"text":"The repository provides whole-human and specialised checkpoints, plus workflows for annotation, integration and perturbation tasks.","source_ids":["src-discovery-bowang-lab-scgpt"],"source_locator":"README.md: Pretrained scGPT checkpoints and Tutorials; scgpt/model/model.py at cebd6fae655b9c585a4807daa3ac31bb764f06b4, lines 79–188"}],"limitations":[{"text":"Whole-human, organ-specific and continually pretrained checkpoints are different configurations. A pretraining claim does not establish transfer performance in a new cell population.","source_ids":["src-discovery-bowang-lab-scgpt"],"source_locator":"README.md: Pretrained scGPT checkpoints and Tutorials; scgpt/model/model.py at cebd6fae655b9c585a4807daa3ac31bb764f06b4, lines 79–188"}],"diagram":{"title":"Conceptual procedure","steps":["Gene IDs and expression","Gene / value encoders","Transformer","Cell and gene representations","Adapted task output"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-bowang-lab-scgpt"],"source_locator":"README.md: Pretrained scGPT checkpoints and Tutorials; scgpt/model/model.py at cebd6fae655b9c585a4807daa3ac31bb764f06b4, lines 79–188"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}}} {"id":"catalog-model-scvi","kind":"model","name":"scVI","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["cells-tissues"],"method_types":["baseline"]},"source_ids":["catalog-source-scvi"],"links":[],"attributes":{"entity_level":"family","version":"scvi-tools","reported_name":"scVI","access":"Public software; train a task-specific model on the permitted split.","method_type":"baseline","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"scVI is a candidate method in the cells-tissues catalogue. The linked resource identifies the project; a checkpoint-level profile has not yet been extracted.","sections":[{"title":"Available evidence","body":"The discovery record links to scVI official resource. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["catalog-source-scvi"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"facts":[{"label":"Recorded configuration","value":"scvi-tools","source_ids":["catalog-source-scvi"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["catalog-source-scvi"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"coverage":"limited","gaps":["The linked official resource has not yielded a reviewed architecture or procedure for the configuration 'scvi-tools' in this release.","Unresolved registry fields: checkpoint revision (not yet extracted), training data (not yet extracted), licence (not yet extracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"catalog-model-spliceai","kind":"model","name":"SpliceAI","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["dna-genomes"],"method_types":["specialist"]},"source_ids":["catalog-source-spliceai"],"links":[{"relation":"variant_of","target_id":"discovery-model-spliceai"}],"attributes":{"entity_level":"family","version":"1.3.1","reported_name":"SpliceAI","access":"Public archived code under PolyForm Strict; model weights are CC BY-NC 4.0 for non-commercial use.","method_type":"specialist","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"SpliceAI predicts how sequence variants alter splice-site usage. Its variant annotation tool combines reference sequence with gene annotation.","sections":[{"title":"Architecture","body":"SpliceAI uses a residual convolutional network with dilated filters to integrate sequence context. It predicts donor, acceptor and non-splice-site probabilities along the sequence; comparing alleles converts those predictions into variant scores. The output is not tissue-specific.","source_ids":["src-discovery-illumina-spliceai","src-discovery-tkzeng-pangolin"],"source_locator":"SpliceAI README and linked Jaganathan et al. paper; Pangolin primary paper https://doi.org/10.1186/s13059-022-02664-4, Methods: Deep neural network architecture, direct comparison with SpliceAI"},{"title":"How it works","body":"The tool evaluates reference and alternative alleles and reports predicted acceptor/donor gains and losses with their relative positions. Annotation, search distance and masking determine the reported variant scores.","source_ids":["src-discovery-illumina-spliceai"],"source_locator":"README.md: package description, License, Usage and Frequently Asked Questions"}],"facts":[{"label":"Inputs","value":"VCF, reference FASTA and gene annotation","source_ids":["src-discovery-illumina-spliceai"],"source_locator":"README.md: package description, License, Usage and Frequently Asked Questions"},{"label":"Outputs","value":"Acceptor/donor gain and loss delta scores","source_ids":["src-discovery-illumina-spliceai"],"source_locator":"README.md: package description, License, Usage and Frequently Asked Questions"},{"label":"Configuration in this record","value":"1.3.1","source_ids":["src-discovery-illumina-spliceai"],"source_locator":"README.md: package description, License, Usage and Frequently Asked Questions"}],"strengths":[{"text":"Produces splice-specific scores and predicted event positions without fitting a classifier to the user’s assay labels.","source_ids":["src-discovery-illumina-spliceai"],"source_locator":"README.md: package description, License, Usage and Frequently Asked Questions"}],"limitations":[{"text":"Code, trained models and precomputed annotations have distinct use terms. Annotation and sequence checks can leave variants unscored.","source_ids":["src-discovery-illumina-spliceai"],"source_locator":"README.md: package description, License, Usage and Frequently Asked Questions"}],"diagram":{"title":"Conceptual procedure","steps":["Reference / alternate DNA","Dilated residual convolutions","Acceptor / donor probabilities","Allelic difference","Variant delta scores"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-illumina-spliceai","src-discovery-tkzeng-pangolin"],"source_locator":"SpliceAI README and linked Jaganathan et al. paper; Pangolin primary paper https://doi.org/10.1186/s13059-022-02664-4, Methods: Deep neural network architecture, direct comparison with SpliceAI"},"coverage":"reviewed","gaps":["An immutable hash for the exact 1.3.1 weight files has not been attached to this family record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}}} {"id":"catalog-model-vina","kind":"model","name":"AutoDock Vina","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["molecular-interactions"],"method_types":["baseline"]},"source_ids":["catalog-source-vina"],"links":[],"attributes":{"entity_level":"family","version":"1.2.7","reported_name":"AutoDock Vina","access":"Public docking software; receptor and ligand preparation required.","method_type":"baseline","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"AutoDock Vina is a conventional docking engine for searching ligand conformations against a receptor.","sections":[{"title":"How it works","body":"A scoring function guides gradient-based conformational search. Candidate poses are ranked under the selected docking configuration; receptor preparation and search settings are part of the method.","source_ids":["src-discovery-ccsb-scripps-autodock-vina"],"source_locator":"README.md: AutoDock Vina description and Features"}],"facts":[{"label":"Method class","value":"Docking search with a scoring function","source_ids":["src-discovery-ccsb-scripps-autodock-vina"],"source_locator":"README.md: AutoDock Vina description and Features"},{"label":"Configuration in this record","value":"1.2.7","source_ids":["src-discovery-ccsb-scripps-autodock-vina"],"source_locator":"README.md: AutoDock Vina description and Features"}],"strengths":[{"text":"Provides a procedural comparator with batch and multiple-ligand workflows.","source_ids":["src-discovery-ccsb-scripps-autodock-vina"],"source_locator":"README.md: AutoDock Vina description and Features"}],"limitations":[{"text":"A docking score is not an experimentally measured affinity. Pose and affinity tasks require separate evaluation.","source_ids":["src-discovery-ccsb-scripps-autodock-vina"],"source_locator":"README.md: AutoDock Vina description and Features"}],"diagram":{"title":"Conceptual procedure","steps":["Prepared receptor and ligand","Search configuration","Conformation search","Scoring function","Ranked docking poses"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-ccsb-scripps-autodock-vina"],"source_locator":"README.md: AutoDock Vina description and Features"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}}} {"id":"catalog-source-alphafold-3-server","kind":"source","name":"AlphaFold 3 Server official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://alphafoldserver.com/output-terms","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-alphagenome","kind":"source","name":"AlphaGenome official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/google-deepmind/alphagenome_research","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-boltz-2","kind":"source","name":"Boltz-2 official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/jwohlwend/boltz","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-chai-1","kind":"source","name":"Chai-1 official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/chaidiscovery/chai-lab","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-diffdock-l","kind":"source","name":"DiffDock-L official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/gcorso/DiffDock","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-dnabert-2","kind":"source","name":"DNABERT-2 official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://huggingface.co/zhihan1996/DNABERT-2-117M","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-esm-2","kind":"source","name":"ESM-2 official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/facebookresearch/esm","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-esmfold","kind":"source","name":"ESMFold official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/facebookresearch/esm","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-evo-2","kind":"source","name":"Evo 2 official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/ArcInstitute/evo2","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-gears","kind":"source","name":"GEARS official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/snap-stanford/GEARS","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-geneformer","kind":"source","name":"Geneformer official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://huggingface.co/ctheodoris/Geneformer","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-kraken2","kind":"source","name":"Kraken2 official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/DerrickWood/kraken2","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-metagene-1","kind":"source","name":"METAGENE-1 official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://huggingface.co/metagene-ai/METAGENE-1","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-metaphlan","kind":"source","name":"MetaPhlAn official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/biobakery/MetaPhlAn","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-mimic","kind":"source","name":"MIMIC official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://huggingface.co/polymathic-ai/MIMIC","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-mrna-fm","kind":"source","name":"mRNA-FM official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/ml4bio/RNA-FM","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-nt-v2","kind":"source","name":"Nucleotide Transformer v2 official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://huggingface.co/InstaDeepAI/nucleotide-transformer-v2-50m-multi-species","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-pangolin","kind":"source","name":"Pangolin official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/tkzeng/Pangolin","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-prokbert","kind":"source","name":"ProkBERT official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/nbrg-ppcu/prokbert","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-proteinmpnn","kind":"source","name":"ProteinMPNN official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/dauparas/ProteinMPNN","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-rhofold","kind":"source","name":"RhoFold+ official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/ml4bio/RhoFold","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-rna-fm","kind":"source","name":"RNA-FM official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/ml4bio/RNA-FM","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-scfoundation","kind":"source","name":"scFoundation official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/biomap-research/scFoundation","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-scgpt","kind":"source","name":"scGPT official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/bowang-lab/scGPT","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-scvi","kind":"source","name":"scVI official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/scverse/scvi-tools","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-spliceai","kind":"source","name":"SpliceAI official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/Illumina/SpliceAI","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-vina","kind":"source","name":"AutoDock Vina official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/ccsb-scripps/AutoDock-Vina","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-task-cell-batch-integration","kind":"benchmark","name":"Batch integration","description":"","status":"discovered","facets":{"areas":["cells-tissues"]},"source_ids":["catalog-source-scgpt","catalog-source-scvi"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"Batch integration","scope_note":"Test whether cell identity is retained across donors and batches.","missing_metadata":{"protocol_version":"not_yet_extracted"},"profile":{"summary":"Batch integration asks whether data from different experiments can be combined without losing biological differences.","sections":[{"title":"Procedure","body":"Evaluate both sides of the problem: cells should no longer separate mainly by technical batch, while cell identities and meaningful variation remain distinguishable. scIB provides metrics for these two objectives. Open Problems publishes a concrete batch-integration task; that task and the scIB software are resources, not interchangeable protocol identities.","source_ids":["src-discovery-theislab-scib","src-discovery-openproblems-bio-openproblems"],"source_locator":"scIB README: Metrics; Resources; Open Problems benchmark directory: Batch Integration (https://openproblems.bio/benchmarks/)"}],"facts":[{"label":"Record type","value":"Generic biological task","source_ids":["src-discovery-theislab-scib","src-discovery-openproblems-bio-openproblems"],"source_locator":"scIB README: Metrics; Resources; Open Problems benchmark directory: Batch Integration (https://openproblems.bio/benchmarks/)"},{"label":"Inputs","value":"Annotated single-cell data with batch labels and biological information","source_ids":["src-discovery-theislab-scib","src-discovery-openproblems-bio-openproblems"],"source_locator":"scIB README: Metrics; Resources; Open Problems benchmark directory: Batch Integration (https://openproblems.bio/benchmarks/)"},{"label":"Assessment","value":"Separate biological-conservation and batch-removal assessments","source_ids":["src-discovery-theislab-scib","src-discovery-openproblems-bio-openproblems"],"source_locator":"scIB README: Metrics; Resources; Open Problems benchmark directory: Batch Integration (https://openproblems.bio/benchmarks/)"},{"label":"Biological conservation examples","value":"Cell-type silhouette, ARI/NMI, isolated-label and trajectory conservation metrics","source_ids":["src-discovery-theislab-scib"],"source_locator":"scIB README: Metrics / Biological Conservation"},{"label":"Batch-removal examples","value":"Batch silhouette, graph iLISI, kBET and graph connectivity","source_ids":["src-discovery-theislab-scib"],"source_locator":"scIB README: Metrics / Batch Correction"},{"label":"Comparator guidance","value":"scIB documents established integration methods including Harmony, MNN and scVI; the choice must match the task inputs.","source_ids":["src-discovery-theislab-scib"],"source_locator":"scIB README: Integration Tools"}],"strengths":[{"text":"A two-part assessment exposes overcorrection that a batch-mixing score alone would miss.","source_ids":["src-discovery-theislab-scib","src-discovery-openproblems-bio-openproblems"],"source_locator":"scIB README: Metrics; Resources; Open Problems benchmark directory: Batch Integration (https://openproblems.bio/benchmarks/)"}],"limitations":[{"text":"This task record has no single dataset or executable split. No run is implied by its relationship to scIB or Open Problems.","source_ids":["src-discovery-theislab-scib","src-discovery-openproblems-bio-openproblems"],"source_locator":"scIB README: Metrics; Resources; Open Problems benchmark directory: Batch Integration (https://openproblems.bio/benchmarks/)"}],"diagram":{"title":"Procedure overview","steps":["Annotated cell data","Apply integration method","Check retained biology","Check reduced batch effects"],"caption":"Conceptual overview, not an executable specification.","source_ids":["src-discovery-theislab-scib","src-discovery-openproblems-bio-openproblems"],"source_locator":"scIB README: Metrics; Resources; Open Problems benchmark directory: Batch Integration (https://openproblems.bio/benchmarks/)"},"coverage":"reviewed","gaps":["Select a concrete dataset, integration output type and metric implementation for any future evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is profile review, not independent execution or numerical reproduction."}}}} {"id":"catalog-task-cell-perturbation","kind":"benchmark","name":"Perturbation response","description":"","status":"discovered","facets":{"areas":["cells-tissues"]},"source_ids":["catalog-source-geneformer","catalog-source-scfoundation","catalog-source-gears"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"Perturbation response","scope_note":"Predict expression changes after unseen perturbations.","missing_metadata":{"protocol_version":"not_yet_extracted"},"profile":{"summary":"How does gene expression change after an intervention?","sections":[{"title":"Proposed comparison design","body":"A proposed evaluation must specify perturbation identity, cell context, dose, time and which perturbations or contexts are withheld.","source_ids":["catalog-source-geneformer","catalog-source-scfoundation","catalog-source-gears"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"facts":[{"label":"Record type","value":"Generic task; proposed design","source_ids":["catalog-source-geneformer","catalog-source-scfoundation","catalog-source-gears"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Inputs","value":"Predict expression changes after unseen perturbations.","source_ids":["catalog-source-geneformer","catalog-source-scfoundation","catalog-source-gears"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Assessment","value":"No single dataset, split or scoring implementation is fixed by this task record","source_ids":["catalog-source-geneformer","catalog-source-scfoundation","catalog-source-gears"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"strengths":[{"text":"No-change and observed-control responses are useful candidate references, not measured results here.","source_ids":["catalog-source-geneformer","catalog-source-scfoundation","catalog-source-gears"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"limitations":[{"text":"A mean expression prediction can hide differential responses; a dataset and metric protocol are still required.","source_ids":["catalog-source-geneformer","catalog-source-scfoundation","catalog-source-gears"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"diagram":{"title":"Proposed evaluation design","steps":["Define cell context and intervention","Withhold perturbations or contexts","Predict expression response","Compare measured response"],"caption":"Conceptual design only. This generic task has no fixed executable protocol and does not imply a completed run.","source_ids":["catalog-source-geneformer","catalog-source-scfoundation","catalog-source-gears"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},"coverage":"limited","gaps":["A concrete protocol, dataset release, split manifest and metric implementation must be selected and source-checked.","Candidate comparisons in this guide are proposals, not recorded evaluations."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"catalog-task-cell-reference-mapping","kind":"benchmark","name":"Donor-held-out reference mapping","description":"","status":"discovered","facets":{"areas":["cells-tissues"]},"source_ids":["catalog-source-geneformer","catalog-source-scgpt","catalog-source-scfoundation","catalog-source-scvi"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"Donor-held-out reference mapping","scope_note":"Map unseen donors to a labelled cell-type reference.","missing_metadata":{"protocol_version":"not_yet_extracted"},"profile":{"summary":"Can cells from an unseen donor be assigned to the correct reference cell types?","sections":[{"title":"Proposed comparison design","body":"Build a labelled reference and keep query donors outside downstream fitting. Record whether novel cell types can be rejected instead of forced into a known label.","source_ids":["catalog-source-geneformer","catalog-source-scgpt","catalog-source-scfoundation","catalog-source-scvi"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"facts":[{"label":"Record type","value":"Generic task; proposed design","source_ids":["catalog-source-geneformer","catalog-source-scgpt","catalog-source-scfoundation","catalog-source-scvi"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Inputs","value":"Map unseen donors to a labelled cell-type reference.","source_ids":["catalog-source-geneformer","catalog-source-scgpt","catalog-source-scfoundation","catalog-source-scvi"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Assessment","value":"No single dataset, split or scoring implementation is fixed by this task record","source_ids":["catalog-source-geneformer","catalog-source-scgpt","catalog-source-scfoundation","catalog-source-scvi"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"strengths":[{"text":"Reference-only nearest-neighbour mapping is a candidate comparison when the input representation is matched.","source_ids":["catalog-source-geneformer","catalog-source-scgpt","catalog-source-scfoundation","catalog-source-scvi"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"limitations":[{"text":"Randomly withholding cells from the same donor would answer a different generalization question.","source_ids":["catalog-source-geneformer","catalog-source-scgpt","catalog-source-scfoundation","catalog-source-scvi"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"diagram":{"title":"Proposed evaluation design","steps":["Label reference cells","Withhold query donors","Map query cells to reference","Check labels and unknown types"],"caption":"Conceptual design only. This generic task has no fixed executable protocol and does not imply a completed run.","source_ids":["catalog-source-geneformer","catalog-source-scgpt","catalog-source-scfoundation","catalog-source-scvi"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},"coverage":"limited","gaps":["A concrete protocol, dataset release, split manifest and metric implementation must be selected and source-checked.","Candidate comparisons in this guide are proposals, not recorded evaluations."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"catalog-task-community-profiling","kind":"benchmark","name":"Community profiling","description":"","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":["catalog-source-metaphlan"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"Community profiling","scope_note":"Estimate taxon abundances in metagenomic samples.","missing_metadata":{"protocol_version":"not_yet_extracted"},"profile":{"summary":"Which microbial taxa occur in a sample, and at what abundance?","sections":[{"title":"Proposed comparison design","body":"A comparison needs matched sample definitions, reference taxonomy versions and a declared abundance convention.","source_ids":["catalog-source-metaphlan"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"facts":[{"label":"Record type","value":"Generic task; proposed design","source_ids":["catalog-source-metaphlan"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Inputs","value":"Estimate taxon abundances in metagenomic samples.","source_ids":["catalog-source-metaphlan"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Assessment","value":"No single dataset, split or scoring implementation is fixed by this task record","source_ids":["catalog-source-metaphlan"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"strengths":[{"text":"Taxon presence and abundance errors should be inspected separately.","source_ids":["catalog-source-metaphlan"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"limitations":[{"text":"Relative abundances depend on the reference and measurement process; this record does not define one mock community.","source_ids":["catalog-source-metaphlan"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"diagram":{"title":"Proposed evaluation design","steps":["Define community samples","Pin reference taxonomy","Estimate taxon abundances","Compare presence and abundance"],"caption":"Conceptual design only. This generic task has no fixed executable protocol and does not imply a completed run.","source_ids":["catalog-source-metaphlan"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},"coverage":"limited","gaps":["A concrete protocol, dataset release, split manifest and metric implementation must be selected and source-checked.","Candidate comparisons in this guide are proposals, not recorded evaluations."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"catalog-task-complex-structure","kind":"benchmark","name":"Biomolecular complex structure","description":"","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":["catalog-source-chai-1","catalog-source-boltz-2","catalog-source-alphafold-3-server"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"Biomolecular complex structure","scope_note":"Predict joint structure for interacting proteins and other molecules.","missing_metadata":{"protocol_version":"not_yet_extracted"},"profile":{"summary":"What is the joint three-dimensional arrangement of interacting molecules?","sections":[{"title":"Proposed comparison design","body":"Specify all chains and molecular partners, permitted templates and alignments, sampling budget and the rule for choosing the reported structure.","source_ids":["catalog-source-chai-1","catalog-source-boltz-2","catalog-source-alphafold-3-server"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"facts":[{"label":"Record type","value":"Generic task; proposed design","source_ids":["catalog-source-chai-1","catalog-source-boltz-2","catalog-source-alphafold-3-server"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Inputs","value":"Predict joint structure for interacting proteins and other molecules.","source_ids":["catalog-source-chai-1","catalog-source-boltz-2","catalog-source-alphafold-3-server"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Assessment","value":"No single dataset, split or scoring implementation is fixed by this task record","source_ids":["catalog-source-chai-1","catalog-source-boltz-2","catalog-source-alphafold-3-server"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"strengths":[{"text":"Interface and whole-complex assessments describe different aspects of structural quality.","source_ids":["catalog-source-chai-1","catalog-source-boltz-2","catalog-source-alphafold-3-server"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"limitations":[{"text":"An accurate monomer fold does not establish that its interaction interface or binding partner is correct.","source_ids":["catalog-source-chai-1","catalog-source-boltz-2","catalog-source-alphafold-3-server"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"diagram":{"title":"Proposed evaluation design","steps":["Specify molecular partners","Fix templates and alignments","Predict and select complex","Assess fold and interface"],"caption":"Conceptual design only. This generic task has no fixed executable protocol and does not imply a completed run.","source_ids":["catalog-source-chai-1","catalog-source-boltz-2","catalog-source-alphafold-3-server"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},"coverage":"limited","gaps":["A concrete protocol, dataset release, split manifest and metric implementation must be selected and source-checked.","Candidate comparisons in this guide are proposals, not recorded evaluations."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"catalog-task-enhancer-effects","kind":"benchmark","name":"Enhancer / MPRA effects","description":"","status":"discovered","facets":{"areas":["dna-genomes"]},"source_ids":["catalog-source-dnabert-2","catalog-source-nt-v2","catalog-source-evo-2","catalog-source-alphagenome"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"Enhancer / MPRA effects","scope_note":"Predict measured activity changes from regulatory sequence variants.","missing_metadata":{"protocol_version":"not_yet_extracted"},"profile":{"summary":"Do regulatory variants change measured enhancer activity?","sections":[{"title":"Proposed comparison design","body":"Select a specific reporter or endogenous assay, pair reference and alternate sequences and withhold the relevant loci or experimental groups.","source_ids":["catalog-source-dnabert-2","catalog-source-nt-v2","catalog-source-evo-2","catalog-source-alphagenome"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"facts":[{"label":"Record type","value":"Generic task; proposed design","source_ids":["catalog-source-dnabert-2","catalog-source-nt-v2","catalog-source-evo-2","catalog-source-alphagenome"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Inputs","value":"Predict measured activity changes from regulatory sequence variants.","source_ids":["catalog-source-dnabert-2","catalog-source-nt-v2","catalog-source-evo-2","catalog-source-alphagenome"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Assessment","value":"No single dataset, split or scoring implementation is fixed by this task record","source_ids":["catalog-source-dnabert-2","catalog-source-nt-v2","catalog-source-evo-2","catalog-source-alphagenome"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"strengths":[{"text":"A measured activity endpoint can be compared with simple sequence-based references.","source_ids":["catalog-source-dnabert-2","catalog-source-nt-v2","catalog-source-evo-2","catalog-source-alphagenome"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"limitations":[{"text":"Reporter activity and endogenous regulation have different contexts; neither implies a universal clinical interpretation.","source_ids":["catalog-source-dnabert-2","catalog-source-nt-v2","catalog-source-evo-2","catalog-source-alphagenome"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"diagram":{"title":"Proposed evaluation design","steps":["Choose regulatory assay","Pair reference and variant","Predict activity change","Compare measured effects"],"caption":"Conceptual design only. This generic task has no fixed executable protocol and does not imply a completed run.","source_ids":["catalog-source-dnabert-2","catalog-source-nt-v2","catalog-source-evo-2","catalog-source-alphagenome"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},"coverage":"limited","gaps":["A concrete protocol, dataset release, split manifest and metric implementation must be selected and source-checked.","Candidate comparisons in this guide are proposals, not recorded evaluations."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"catalog-task-heldout-clade","kind":"benchmark","name":"Held-out-clade classification","description":"","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":["catalog-source-prokbert","catalog-source-metagene-1","catalog-source-evo-2","catalog-source-kraken2"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"Held-out-clade classification","scope_note":"Hold clades out of downstream fitting and reference databases; evaluate known ancestor labels or unknown-taxon detection, and audit pretraining overlap separately.","missing_metadata":{"protocol_version":"not_yet_extracted"},"profile":{"summary":"Can a sequence method generalize beyond taxa present during fitting?","sections":[{"title":"Proposed comparison design","body":"Withhold the chosen clades from downstream training and reference databases, then separately audit overlap with model pretraining.","source_ids":["catalog-source-prokbert","catalog-source-metagene-1","catalog-source-evo-2","catalog-source-kraken2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"facts":[{"label":"Record type","value":"Generic task; proposed design","source_ids":["catalog-source-prokbert","catalog-source-metagene-1","catalog-source-evo-2","catalog-source-kraken2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Inputs","value":"Hold clades out of downstream fitting and reference databases; evaluate known ancestor labels or unknown-taxon detection, and audit pretraining overlap separately.","source_ids":["catalog-source-prokbert","catalog-source-metagene-1","catalog-source-evo-2","catalog-source-kraken2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Assessment","value":"No single dataset, split or scoring implementation is fixed by this task record","source_ids":["catalog-source-prokbert","catalog-source-metagene-1","catalog-source-evo-2","catalog-source-kraken2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"strengths":[{"text":"Taxonomic distance makes the extrapolation challenge explicit.","source_ids":["catalog-source-prokbert","catalog-source-metagene-1","catalog-source-evo-2","catalog-source-kraken2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"limitations":[{"text":"Withholding species is not the same as withholding their genera; pretraining overlap is a separate unresolved question.","source_ids":["catalog-source-prokbert","catalog-source-metagene-1","catalog-source-evo-2","catalog-source-kraken2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"diagram":{"title":"Proposed evaluation design","steps":["Choose withheld clades","Audit training and references","Classify held-out sequences","Assess ancestral or unknown labels"],"caption":"Conceptual design only. This generic task has no fixed executable protocol and does not imply a completed run.","source_ids":["catalog-source-prokbert","catalog-source-metagene-1","catalog-source-evo-2","catalog-source-kraken2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},"coverage":"limited","gaps":["A concrete protocol, dataset release, split manifest and metric implementation must be selected and source-checked.","Candidate comparisons in this guide are proposals, not recorded evaluations."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"catalog-task-ligand-affinity","kind":"benchmark","name":"Small-molecule affinity","description":"","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":["catalog-source-boltz-2"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"Small-molecule affinity","scope_note":"Predict measured binding affinity; pose confidence is not an affinity value.","missing_metadata":{"protocol_version":"not_yet_extracted"},"profile":{"summary":"How strongly does a molecule bind its target under the measured conditions?","sections":[{"title":"Proposed comparison design","body":"Select a defined affinity assay and unit, separate compounds and targets according to the intended transfer test, and record any structural information supplied.","source_ids":["catalog-source-boltz-2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"facts":[{"label":"Record type","value":"Generic task; proposed design","source_ids":["catalog-source-boltz-2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Inputs","value":"Predict measured binding affinity; pose confidence is not an affinity value.","source_ids":["catalog-source-boltz-2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Assessment","value":"No single dataset, split or scoring implementation is fixed by this task record","source_ids":["catalog-source-boltz-2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"strengths":[{"text":"Measured affinities provide a quantitative endpoint when assays are comparable.","source_ids":["catalog-source-boltz-2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"limitations":[{"text":"A pose-confidence score is not an affinity; Kd, Ki and activity measurements must not be silently combined.","source_ids":["catalog-source-boltz-2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"diagram":{"title":"Proposed evaluation design","steps":["Choose affinity assay","Define compound and target split","Predict binding strength","Compare matched affinity units"],"caption":"Conceptual design only. This generic task has no fixed executable protocol and does not imply a completed run.","source_ids":["catalog-source-boltz-2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},"coverage":"limited","gaps":["A concrete protocol, dataset release, split manifest and metric implementation must be selected and source-checked.","Candidate comparisons in this guide are proposals, not recorded evaluations."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"catalog-task-ligand-pose","kind":"benchmark","name":"Protein–ligand pose","description":"","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":["catalog-source-chai-1","catalog-source-boltz-2","catalog-source-diffdock-l","catalog-source-alphafold-3-server","catalog-source-vina"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"Protein–ligand pose","scope_note":"Predict the bound ligand geometry from prepared molecular inputs.","missing_metadata":{"protocol_version":"not_yet_extracted"},"profile":{"summary":"Where and how does a ligand sit in a protein binding site?","sections":[{"title":"Proposed comparison design","body":"Declare whether the pocket is known, how the receptor and ligand are prepared, and how one pose is selected from sampled candidates.","source_ids":["catalog-source-chai-1","catalog-source-boltz-2","catalog-source-diffdock-l","catalog-source-alphafold-3-server","catalog-source-vina"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"facts":[{"label":"Record type","value":"Generic task; proposed design","source_ids":["catalog-source-chai-1","catalog-source-boltz-2","catalog-source-diffdock-l","catalog-source-alphafold-3-server","catalog-source-vina"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Inputs","value":"Predict the bound ligand geometry from prepared molecular inputs.","source_ids":["catalog-source-chai-1","catalog-source-boltz-2","catalog-source-diffdock-l","catalog-source-alphafold-3-server","catalog-source-vina"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Assessment","value":"No single dataset, split or scoring implementation is fixed by this task record","source_ids":["catalog-source-chai-1","catalog-source-boltz-2","catalog-source-diffdock-l","catalog-source-alphafold-3-server","catalog-source-vina"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"strengths":[{"text":"Geometry validity and agreement with a reference pose capture complementary failure modes.","source_ids":["catalog-source-chai-1","catalog-source-boltz-2","catalog-source-diffdock-l","catalog-source-alphafold-3-server","catalog-source-vina"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"limitations":[{"text":"A best-of-many pose chosen using the reference cannot be compared with a prospectively selected top-ranked pose.","source_ids":["catalog-source-chai-1","catalog-source-boltz-2","catalog-source-diffdock-l","catalog-source-alphafold-3-server","catalog-source-vina"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"diagram":{"title":"Proposed evaluation design","steps":["Prepare receptor and ligand","Declare pocket information","Generate and select poses","Assess geometry and validity"],"caption":"Conceptual design only. This generic task has no fixed executable protocol and does not imply a completed run.","source_ids":["catalog-source-chai-1","catalog-source-boltz-2","catalog-source-diffdock-l","catalog-source-alphafold-3-server","catalog-source-vina"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},"coverage":"limited","gaps":["A concrete protocol, dataset release, split manifest and metric implementation must be selected and source-checked.","Candidate comparisons in this guide are proposals, not recorded evaluations."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"catalog-task-long-range-regulation","kind":"benchmark","name":"Long-range regulation","description":"","status":"discovered","facets":{"areas":["dna-genomes"]},"source_ids":["catalog-source-evo-2","catalog-source-alphagenome"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"Long-range regulation","scope_note":"Predict gene-expression or chromatin effects from long-context DNA sequence.","missing_metadata":{"protocol_version":"not_yet_extracted"},"profile":{"summary":"Can extended genomic context help predict regulatory effects?","sections":[{"title":"Proposed comparison design","body":"A protocol must identify the genome assembly, genomic interval, input length and measured expression or chromatin endpoint.","source_ids":["catalog-source-evo-2","catalog-source-alphagenome"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"facts":[{"label":"Record type","value":"Generic task; proposed design","source_ids":["catalog-source-evo-2","catalog-source-alphagenome"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Inputs","value":"Predict gene-expression or chromatin effects from long-context DNA sequence.","source_ids":["catalog-source-evo-2","catalog-source-alphagenome"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Assessment","value":"No single dataset, split or scoring implementation is fixed by this task record","source_ids":["catalog-source-evo-2","catalog-source-alphagenome"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"strengths":[{"text":"Explicit context length makes information available to each model inspectable.","source_ids":["catalog-source-evo-2","catalog-source-alphagenome"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"limitations":[{"text":"Different context windows and experimental targets can turn superficially similar scores into different tasks.","source_ids":["catalog-source-evo-2","catalog-source-alphagenome"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"diagram":{"title":"Proposed evaluation design","steps":["Pin genome and sequence window","Define held-out loci","Predict regulatory endpoint","Compare experimental signal"],"caption":"Conceptual design only. This generic task has no fixed executable protocol and does not imply a completed run.","source_ids":["catalog-source-evo-2","catalog-source-alphagenome"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},"coverage":"limited","gaps":["A concrete protocol, dataset release, split manifest and metric implementation must be selected and source-checked.","Candidate comparisons in this guide are proposals, not recorded evaluations."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"catalog-task-mfass-splice","kind":"benchmark","name":"MFASS splice-variant prioritisation","description":"","status":"discovered","facets":{"areas":["dna-genomes"]},"source_ids":["catalog-source-dnabert-2","catalog-source-nt-v2","catalog-source-spliceai","catalog-source-pangolin"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"MFASS splice-variant prioritisation","scope_note":"Functional exon-recognition assay; mfass-v2 reports a corrected baseline and one complete local DNABERT-2 protocol.","missing_metadata":{"protocol_version":"not_yet_extracted"},"profile":{"summary":"MFASS prioritisation tests whether model scores enrich for experimentally disrupted exon recognition.","sections":[{"title":"Procedure","body":"The concrete rewire v2 protocol validates assay-oriented sequence pairs and evaluates a predeclared grouped split. This catalogue task describes the capability; it is distinct from the assay dataset and the pinned v2 implementation.","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/README.md at bee9133b83f3aedaf2bbb9013f1875515845607e: Correction; Dataset; Cohort reconciliation; Split; results JSON in benchmarks/mfass/results/"}],"facts":[{"label":"Record type","value":"Task linked to a concrete protocol","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/README.md at bee9133b83f3aedaf2bbb9013f1875515845607e: Correction; Dataset; Cohort reconciliation; Split; results JSON in benchmarks/mfass/results/"},{"label":"Inputs","value":"Functional reporter labels and variant scores","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/README.md at bee9133b83f3aedaf2bbb9013f1875515845607e: Correction; Dataset; Cohort reconciliation; Split; results JSON in benchmarks/mfass/results/"},{"label":"Assessment","value":"Top-ranked assay positives and ranking metrics","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/README.md at bee9133b83f3aedaf2bbb9013f1875515845607e: Correction; Dataset; Cohort reconciliation; Split; results JSON in benchmarks/mfass/results/"}],"strengths":[{"text":"Experimental reporter labels offer a functional endpoint independent of clinical assertions.","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/README.md at bee9133b83f3aedaf2bbb9013f1875515845607e: Correction; Dataset; Cohort reconciliation; Split; results JSON in benchmarks/mfass/results/"}],"limitations":[{"text":"Reporter disruption is not a clinical diagnosis, and models with different sequence context are not identical-input baselines.","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/README.md at bee9133b83f3aedaf2bbb9013f1875515845607e: Correction; Dataset; Cohort reconciliation; Split; results JSON in benchmarks/mfass/results/"}],"diagram":{"title":"Proposed evaluation design","steps":["Choose matched biological data","Define held-out evaluation","Apply candidate methods","Assess the specified endpoint"],"caption":"Conceptual design only. This generic task has no fixed executable protocol and does not imply a completed run.","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/README.md at bee9133b83f3aedaf2bbb9013f1875515845607e: Correction; Dataset; Cohort reconciliation; Split; results JSON in benchmarks/mfass/results/"},"coverage":"reviewed","gaps":["Use the pinned MFASS v2 protocol rather than this generic task identity when recording a new run."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is profile review, not independent execution or numerical reproduction."}}}} {"id":"catalog-task-microbial-promoters","kind":"benchmark","name":"Bacterial promoter prediction","description":"","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":["catalog-source-prokbert","catalog-source-evo-2"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"Bacterial promoter prediction","scope_note":"Classify promoter activity from microbial DNA sequence.","missing_metadata":{"protocol_version":"not_yet_extracted"},"profile":{"summary":"Does a microbial DNA region act as a promoter in the specified organism and assay?","sections":[{"title":"Proposed comparison design","body":"Define the organism, promoter class, negative sampling strategy and held-out sequence groups before evaluating a classifier.","source_ids":["catalog-source-prokbert","catalog-source-evo-2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"facts":[{"label":"Record type","value":"Generic task; proposed design","source_ids":["catalog-source-prokbert","catalog-source-evo-2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Inputs","value":"Classify promoter activity from microbial DNA sequence.","source_ids":["catalog-source-prokbert","catalog-source-evo-2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Assessment","value":"No single dataset, split or scoring implementation is fixed by this task record","source_ids":["catalog-source-prokbert","catalog-source-evo-2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"strengths":[{"text":"A promoter assay anchors a sequence prediction to a specific regulatory function.","source_ids":["catalog-source-prokbert","catalog-source-evo-2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"limitations":[{"text":"Negative construction and sequence similarity can dominate classification difficulty; no specific split is established here.","source_ids":["catalog-source-prokbert","catalog-source-evo-2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"diagram":{"title":"Proposed evaluation design","steps":["Specify organism and promoter class","Define matched negatives","Predict promoter labels","Score held-out sequence groups"],"caption":"Conceptual design only. This generic task has no fixed executable protocol and does not imply a completed run.","source_ids":["catalog-source-prokbert","catalog-source-evo-2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},"coverage":"limited","gaps":["A concrete protocol, dataset release, split manifest and metric implementation must be selected and source-checked.","Candidate comparisons in this guide are proposals, not recorded evaluations."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"catalog-task-phage-pathogen-reads","kind":"benchmark","name":"Phage / pathogen reads","description":"","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":["catalog-source-prokbert","catalog-source-metagene-1","catalog-source-kraken2"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"Phage / pathogen reads","scope_note":"Classify held-out phage or pathogen sequences and record taxonomic distance.","missing_metadata":{"protocol_version":"not_yet_extracted"},"profile":{"summary":"Can a method recognize phage or pathogen sequences beyond its references?","sections":[{"title":"Proposed comparison design","body":"Record sequence length, sequencing error, taxonomic distance and reference-database cutoff, then assess the intended classification level.","source_ids":["catalog-source-prokbert","catalog-source-metagene-1","catalog-source-kraken2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"facts":[{"label":"Record type","value":"Generic task; proposed design","source_ids":["catalog-source-prokbert","catalog-source-metagene-1","catalog-source-kraken2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Inputs","value":"Classify held-out phage or pathogen sequences and record taxonomic distance.","source_ids":["catalog-source-prokbert","catalog-source-metagene-1","catalog-source-kraken2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Assessment","value":"No single dataset, split or scoring implementation is fixed by this task record","source_ids":["catalog-source-prokbert","catalog-source-metagene-1","catalog-source-kraken2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"strengths":[{"text":"Held-out taxa allow the evaluation to distinguish recognition from close-reference matching.","source_ids":["catalog-source-prokbert","catalog-source-metagene-1","catalog-source-kraken2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"limitations":[{"text":"A positive sequence classification does not establish abundance, infectivity or clinical disease.","source_ids":["catalog-source-prokbert","catalog-source-metagene-1","catalog-source-kraken2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"diagram":{"title":"Proposed evaluation design","steps":["Define reads and reference cutoff","Withhold target taxa","Classify sequence fragments","Report rank and coverage"],"caption":"Conceptual design only. This generic task has no fixed executable protocol and does not imply a completed run.","source_ids":["catalog-source-prokbert","catalog-source-metagene-1","catalog-source-kraken2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},"coverage":"limited","gaps":["A concrete protocol, dataset release, split manifest and metric implementation must be selected and source-checked.","Candidate comparisons in this guide are proposals, not recorded evaluations."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"catalog-task-protein-design","kind":"benchmark","name":"Protein design / inverse folding","description":"","status":"discovered","facets":{"areas":["proteins-complexes"]},"source_ids":["catalog-source-proteinmpnn"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"Protein design / inverse folding","scope_note":"Score or design sequences conditional on a known structure.","missing_metadata":{"protocol_version":"not_yet_extracted"},"profile":{"summary":"Can a method propose sequences compatible with a desired protein structure?","sections":[{"title":"Proposed comparison design","body":"Separate sequence scoring from generation, define the target structure and permitted constraints, and record how designs are selected for evaluation.","source_ids":["catalog-source-proteinmpnn"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"facts":[{"label":"Record type","value":"Generic task; proposed design","source_ids":["catalog-source-proteinmpnn"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Inputs","value":"Score or design sequences conditional on a known structure.","source_ids":["catalog-source-proteinmpnn"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Assessment","value":"No single dataset, split or scoring implementation is fixed by this task record","source_ids":["catalog-source-proteinmpnn"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"strengths":[{"text":"Structure-conditioned design makes the intended molecular target explicit.","source_ids":["catalog-source-proteinmpnn"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"limitations":[{"text":"Sequence recovery or predicted refolding does not establish experimentally measured function.","source_ids":["catalog-source-proteinmpnn"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"diagram":{"title":"Proposed evaluation design","steps":["Specify target structure","Generate candidate sequences","Select designs without test leakage","Evaluate intended property"],"caption":"Conceptual design only. This generic task has no fixed executable protocol and does not imply a completed run.","source_ids":["catalog-source-proteinmpnn"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},"coverage":"limited","gaps":["A concrete protocol, dataset release, split manifest and metric implementation must be selected and source-checked.","Candidate comparisons in this guide are proposals, not recorded evaluations."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"catalog-task-protein-monomer-structure","kind":"benchmark","name":"Monomer structure","description":"","status":"discovered","facets":{"areas":["proteins-complexes"]},"source_ids":["catalog-source-esm-2","catalog-source-esmfold","catalog-source-chai-1"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"Monomer structure","scope_note":"Predict single-chain structure from sequence.","missing_metadata":{"protocol_version":"not_yet_extracted"},"profile":{"summary":"Can a protein sequence be mapped to an accurate single-chain structure?","sections":[{"title":"Proposed comparison design","body":"Specify permitted templates, sequence alignments and structure-release cutoff, then compare predictions with held-out experimental structures.","source_ids":["catalog-source-esm-2","catalog-source-esmfold","catalog-source-chai-1"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"facts":[{"label":"Record type","value":"Generic task; proposed design","source_ids":["catalog-source-esm-2","catalog-source-esmfold","catalog-source-chai-1"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Inputs","value":"Predict single-chain structure from sequence.","source_ids":["catalog-source-esm-2","catalog-source-esmfold","catalog-source-chai-1"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Assessment","value":"No single dataset, split or scoring implementation is fixed by this task record","source_ids":["catalog-source-esm-2","catalog-source-esmfold","catalog-source-chai-1"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"strengths":[{"text":"An explicit reference structure permits local and global geometric assessment.","source_ids":["catalog-source-esm-2","catalog-source-esmfold","catalog-source-chai-1"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"limitations":[{"text":"Single-chain accuracy does not imply correct complexes, dynamics or ligand interactions.","source_ids":["catalog-source-esm-2","catalog-source-esmfold","catalog-source-chai-1"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"diagram":{"title":"Proposed evaluation design","steps":["Specify sequence and allowed inputs","Predict single-chain structure","Match experimental reference","Assess local and global geometry"],"caption":"Conceptual design only. This generic task has no fixed executable protocol and does not imply a completed run.","source_ids":["catalog-source-esm-2","catalog-source-esmfold","catalog-source-chai-1"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},"coverage":"limited","gaps":["A concrete protocol, dataset release, split manifest and metric implementation must be selected and source-checked.","Candidate comparisons in this guide are proposals, not recorded evaluations."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"catalog-task-proteingym-effects","kind":"benchmark","name":"ProteinGym mutation effects","description":"","status":"discovered","facets":{"areas":["proteins-complexes"]},"source_ids":["catalog-source-mimic","catalog-source-esm-2","catalog-source-proteinmpnn"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"ProteinGym mutation effects","scope_note":"Rank substitution effects within held-out deep-mutational-scanning assays.","missing_metadata":{"protocol_version":"not_yet_extracted"},"profile":{"summary":"Protein mutation-effect evaluation asks whether scores order variants consistently with measured assay effects.","sections":[{"title":"Procedure","body":"ProteinGym provides concrete assay releases and evaluation tracks. Select the assay, permitted model inputs and supervision regime before using its scoring and aggregation procedure.","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"ProteinGym README: Overview; Results; Resources"}],"facts":[{"label":"Record type","value":"Generic task linked to a suite","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"ProteinGym README: Overview; Results; Resources"},{"label":"Inputs","value":"Protein variants and experimental assay measurements","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"ProteinGym README: Overview; Results; Resources"},{"label":"Assessment","value":"Within-assay ranking and protocol-specific aggregation","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"ProteinGym README: Overview; Results; Resources"}],"strengths":[{"text":"Assay-level analysis distinguishes effects measured under different biological conditions.","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"ProteinGym README: Overview; Results; Resources"}],"limitations":[{"text":"This generic task does not fix one ProteinGym release or train/test split.","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"ProteinGym README: Overview; Results; Resources"}],"diagram":{"title":"Proposed evaluation design","steps":["Choose matched biological data","Define held-out evaluation","Apply candidate methods","Assess the specified endpoint"],"caption":"Conceptual design only. This generic task has no fixed executable protocol and does not imply a completed run.","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"ProteinGym README: Overview; Results; Resources"},"coverage":"reviewed","gaps":["Select a ProteinGym release, track and assay subset."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is profile review, not independent execution or numerical reproduction."}}}} {"id":"catalog-task-rna-secondary-structure","kind":"benchmark","name":"RNA secondary structure","description":"","status":"discovered","facets":{"areas":["rna-transcriptomes"]},"source_ids":["catalog-source-rna-fm","catalog-source-rhofold"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"RNA secondary structure","scope_note":"Compare predicted base pairs against held-out RNA structures.","missing_metadata":{"protocol_version":"not_yet_extracted"},"profile":{"summary":"Which nucleotides pair within an RNA molecule?","sections":[{"title":"Proposed comparison design","body":"A protocol needs labelled base pairs, a held-out sequence or family split and explicit treatment of pseudoknots and invalid predicted pairs.","source_ids":["catalog-source-rna-fm","catalog-source-rhofold"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"facts":[{"label":"Record type","value":"Generic task; proposed design","source_ids":["catalog-source-rna-fm","catalog-source-rhofold"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Inputs","value":"Compare predicted base pairs against held-out RNA structures.","source_ids":["catalog-source-rna-fm","catalog-source-rhofold"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Assessment","value":"No single dataset, split or scoring implementation is fixed by this task record","source_ids":["catalog-source-rna-fm","catalog-source-rhofold"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"strengths":[{"text":"Base-pair assessment can identify errors hidden by sequence-level summaries.","source_ids":["catalog-source-rna-fm","catalog-source-rhofold"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"limitations":[{"text":"A secondary-structure score does not measure the full three-dimensional fold; RNA-family overlap must be assessed.","source_ids":["catalog-source-rna-fm","catalog-source-rhofold"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"diagram":{"title":"Proposed evaluation design","steps":["Choose RNA sequences and labels","Withhold sequences or families","Predict base pairs","Assess allowed pairing classes"],"caption":"Conceptual design only. This generic task has no fixed executable protocol and does not imply a completed run.","source_ids":["catalog-source-rna-fm","catalog-source-rhofold"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},"coverage":"limited","gaps":["A concrete protocol, dataset release, split manifest and metric implementation must be selected and source-checked.","Candidate comparisons in this guide are proposals, not recorded evaluations."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"catalog-task-rna-splice-sites","kind":"benchmark","name":"RNA splice-site mapping","description":"","status":"discovered","facets":{"areas":["rna-transcriptomes"]},"source_ids":["catalog-source-rna-fm","catalog-source-mimic"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"RNA splice-site mapping","scope_note":"Predict splice-site classes from transcript sequence, using a held-out gene split.","missing_metadata":{"protocol_version":"not_yet_extracted"},"profile":{"summary":"Where are splice donor and acceptor sites in a sequence?","sections":[{"title":"Proposed comparison design","body":"Define the sequence convention, splice-site labels and a held-out gene split before fitting and evaluating the predictor.","source_ids":["catalog-source-rna-fm","catalog-source-mimic"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"facts":[{"label":"Record type","value":"Generic task; proposed design","source_ids":["catalog-source-rna-fm","catalog-source-mimic"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Inputs","value":"Predict splice-site classes from transcript sequence, using a held-out gene split.","source_ids":["catalog-source-rna-fm","catalog-source-mimic"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Assessment","value":"No single dataset, split or scoring implementation is fixed by this task record","source_ids":["catalog-source-rna-fm","catalog-source-mimic"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"strengths":[{"text":"Gene-held-out assessment addresses reuse of closely related transcript context.","source_ids":["catalog-source-rna-fm","catalog-source-mimic"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"limitations":[{"text":"Splice-site recognition and predicting a variant-induced change in splicing are different tasks.","source_ids":["catalog-source-rna-fm","catalog-source-mimic"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"diagram":{"title":"Proposed evaluation design","steps":["Define sequence and site labels","Withhold genes","Predict donor and acceptor sites","Score site recognition"],"caption":"Conceptual design only. This generic task has no fixed executable protocol and does not imply a completed run.","source_ids":["catalog-source-rna-fm","catalog-source-mimic"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},"coverage":"limited","gaps":["A concrete protocol, dataset release, split manifest and metric implementation must be selected and source-checked.","Candidate comparisons in this guide are proposals, not recorded evaluations."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"catalog-task-rna-tertiary-structure","kind":"benchmark","name":"RNA tertiary structure","description":"","status":"discovered","facets":{"areas":["rna-transcriptomes"]},"source_ids":["catalog-source-rna-fm","catalog-source-rhofold"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"RNA tertiary structure","scope_note":"Compare predicted 3D folds against independently held-out structures.","missing_metadata":{"protocol_version":"not_yet_extracted"},"profile":{"summary":"Can a method predict an RNA molecule’s three-dimensional fold?","sections":[{"title":"Proposed comparison design","body":"Record the permitted sequence, secondary-structure and template information; compare predictions with independently held-out structures.","source_ids":["catalog-source-rna-fm","catalog-source-rhofold"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"facts":[{"label":"Record type","value":"Generic task; proposed design","source_ids":["catalog-source-rna-fm","catalog-source-rhofold"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Inputs","value":"Compare predicted 3D folds against independently held-out structures.","source_ids":["catalog-source-rna-fm","catalog-source-rhofold"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Assessment","value":"No single dataset, split or scoring implementation is fixed by this task record","source_ids":["catalog-source-rna-fm","catalog-source-rhofold"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"strengths":[{"text":"Structural references make geometric errors inspectable.","source_ids":["catalog-source-rna-fm","catalog-source-rhofold"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"limitations":[{"text":"Input secondary structure, templates and family similarity change the difficulty; none is fixed by this generic task.","source_ids":["catalog-source-rna-fm","catalog-source-rhofold"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"diagram":{"title":"Proposed evaluation design","steps":["Specify RNA and allowed inputs","Predict three-dimensional fold","Match held-out structure","Assess geometric agreement"],"caption":"Conceptual design only. This generic task has no fixed executable protocol and does not imply a completed run.","source_ids":["catalog-source-rna-fm","catalog-source-rhofold"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},"coverage":"limited","gaps":["A concrete protocol, dataset release, split manifest and metric implementation must be selected and source-checked.","Candidate comparisons in this guide are proposals, not recorded evaluations."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"catalog-task-utr-translation","kind":"benchmark","name":"Translation / RNA stability","description":"","status":"discovered","facets":{"areas":["rna-transcriptomes"]},"source_ids":["catalog-source-mrna-fm","catalog-source-mimic"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"Translation / RNA stability","scope_note":"Predict measured translation or stability effects; choose UTR or coding-sequence assays to match each model’s input modality.","missing_metadata":{"protocol_version":"not_yet_extracted"},"profile":{"summary":"Can sequence predict measured RNA translation or stability?","sections":[{"title":"Proposed comparison design","body":"Choose the relevant UTR or coding-sequence assay and specify its cellular context, sequence length and experimental split.","source_ids":["catalog-source-mrna-fm","catalog-source-mimic"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"facts":[{"label":"Record type","value":"Generic task; proposed design","source_ids":["catalog-source-mrna-fm","catalog-source-mimic"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Inputs","value":"Predict measured translation or stability effects; choose UTR or coding-sequence assays to match each model’s input modality.","source_ids":["catalog-source-mrna-fm","catalog-source-mimic"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Assessment","value":"No single dataset, split or scoring implementation is fixed by this task record","source_ids":["catalog-source-mrna-fm","catalog-source-mimic"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"strengths":[{"text":"A matched assay separates translation and stability endpoints.","source_ids":["catalog-source-mrna-fm","catalog-source-mimic"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"limitations":[{"text":"Ribosome loading, translation efficiency and RNA half-life are different measurements and should not be pooled.","source_ids":["catalog-source-mrna-fm","catalog-source-mimic"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"diagram":{"title":"Proposed evaluation design","steps":["Select matched translation or stability assay","Define held-out sequences","Predict specific assay endpoint","Compare measured response"],"caption":"Conceptual design only. This generic task has no fixed executable protocol and does not imply a completed run.","source_ids":["catalog-source-mrna-fm","catalog-source-mimic"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},"coverage":"limited","gaps":["A concrete protocol, dataset release, split manifest and metric implementation must be selected and source-checked.","Candidate comparisons in this guide are proposals, not recorded evaluations."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"cathe2-2025","kind":"source","name":"CATHe2: Enhanced CATH superfamily detection using ProstT5 and structural alphabets","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["proteins-complexes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12631783/","version":"journal full text in PMC","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.1093/biomethods/bpaf080","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"713dbfb6ec1cc1aa85c0543eb93aafa0b45b8873df28053b765dd0a1b6d9b563","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12631783/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:38.366Z","legacy_paper":{"id":"cathe2-2025","title":"CATHe2: Enhanced CATH superfamily detection using ProstT5 and structural alphabets","year":2025,"publication_status":"peer_reviewed","version":"journal full text in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12631783/","primary_domain":"proteins-complexes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.1093/biomethods/bpaf080","notes":"Numeric result checked against Table 3. in primary full-text XML; journal/source: Biology Methods & Protocols."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"cell-dino-2025","kind":"source","name":"Cell-DINO: Self-supervised image-based embeddings for cell fluorescent microscopy","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["cells-tissues"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12826486/","version":"version of record","retrieved_at":"2026-09-15T23:37:05Z","doi":"10.1371/journal.pcbi.1013828","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"12a53a78c70b3033c3351cf7afd4da42ebc98bb3281308f07e71e5baffc153a0","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12826486/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:06Z","legacy_paper":{"id":"cell-dino-2025","title":"Cell-DINO: Self-supervised image-based embeddings for cell fluorescent microscopy","year":2025,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12826486/","primary_domain":"cells-tissues","retrieved_utc":"2026-09-15T23:37:05Z","notes":"Primary full text verified via Europe PMC fullTextXML; venue: PLOS Computational Biology; PMC ID: PMC12826486. PL column is F1-score reported on a 0–100 scale; Cell-DINO is a vision encoder plus downstream classifier.","doi":"10.1371/journal.pcbi.1013828"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"cell2sentence-2024","kind":"source","name":"Cell2Sentence: Teaching Large Language Models the Language of Biology","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["cells-tissues"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11565894/","version":"preprint archived 2024-10-29","retrieved_at":"2026-09-15T23:25:00Z","doi":"10.1101/2023.09.11.557287","publication_status":"preprint","year":2024,"artifact_sha256":"e088727d6e04857fccb7033a9b074e1850f775e86e7d2e99e603dde09558ab02","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11565894/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:16.533640+00:00","legacy_paper":{"id":"cell2sentence-2024","title":"Cell2Sentence: Teaching Large Language Models the Language of Biology","year":2024,"publication_status":"preprint","version":"preprint archived 2024-10-29","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11565894/","primary_domain":"cells-tissues","retrieved_utc":"2026-09-15T23:25:00Z","notes":"Primary full text via Europe PMC XML; venue: bioRxiv; PMC ID: PMC11565894.","doi":"10.1101/2023.09.11.557287"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"claim-b2-2ome-lm-2025","kind":"claim","name":"Reported AUC for 2OMe-LM","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["human RNA 2-prime-O-methylation site prediction"]},"source_ids":["2ome-lm-2025"],"links":[{"relation":"subject","target_id":"b2-2ome-lm-2025"}],"attributes":{"field":"attributes.printed_value","value":"0.919","source_locator":"Table 1, 2OMe-LM row, AUC column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:38.332Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-b2-antibody-deamidation-plm-2024","kind":"claim","name":"Reported accuracy for ESM-2 650M embeddings + classifier","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["antibody deamidation-site prediction"]},"source_ids":["antibody-deamidation-plm-2024"],"links":[{"relation":"subject","target_id":"b2-antibody-deamidation-plm-2024"}],"attributes":{"field":"attributes.printed_value","value":"0.944","source_locator":"Table 1, Global embeddings only row, Accuracy column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:38.478Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-b2-barcodebert-2026","kind":"claim","name":"Reported accuracy for BarcodeBERT (4–4-4)","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["unseen-species genus classification"]},"source_ids":["barcodebert-2026"],"links":[{"relation":"subject","target_id":"b2-barcodebert-2026"}],"attributes":{"field":"attributes.printed_value","value":"78.5","source_locator":"Table 1, BarcodeBERT (4–4-4) row, unseen-species genus-level 1-NN Acc (%) column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558051+00:00","notes":"Resolved the two-level column header: Acc (%) falls under genus-level 1-NN probe of unseen species, not seen-species classification or BIN reconstruction. BarcodeBERT (4–4-4) has 78.5 in this cell."}}} {"id":"claim-b2-birna-bert-2025","kind":"claim","name":"Reported F1 for BiRNA-BERT","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["extremely long RNA species classification"]},"source_ids":["birna-bert-2025"],"links":[{"relation":"subject","target_id":"b2-birna-bert-2025"}],"attributes":{"field":"attributes.printed_value","value":"0.804","source_locator":"Table 2, BiRNA-BERT row, F1 Score column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:38.292Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-b2-cathe2-2025","kind":"claim","name":"Reported F1 for CATHe2 + ProstT5","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["CATH superfamily annotation"]},"source_ids":["cathe2-2025"],"links":[{"relation":"subject","target_id":"b2-cathe2-2025"}],"attributes":{"field":"attributes.printed_value","value":"82.3","source_locator":"Table 3, ProstT5 full row, F1 score column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:38.366Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-b2-clathrin-plm-2025","kind":"claim","name":"Reported accuracy for ESM-2 embedding + paper classifier","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["clathrin protein classification"]},"source_ids":["clathrin-plm-2025"],"links":[{"relation":"subject","target_id":"b2-clathrin-plm-2025"}],"attributes":{"field":"attributes.printed_value","value":"0.916","source_locator":"Table 2, Independent test / ESM-2 row, ACC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558194+00:00","notes":"Resolved the blank evaluation-strategy cells by their independent-test row group. ESM-2 ACC is 0.916 there; the cross-validation ESM-2 ACC is instead 0.873. This is the paper classifier using embeddings, not a standalone checkpoint."}}} {"id":"claim-b2-cobra-rna-binding-2026","kind":"claim","name":"Reported MCC for ERNIE-RNA + CoBRA","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA compound-binding site prediction"]},"source_ids":["cobra-rna-binding-2026"],"links":[{"relation":"subject","target_id":"b2-cobra-rna-binding-2026"}],"attributes":{"field":"attributes.printed_value","value":"0.657","source_locator":"Table 2, ERNIE-RNA / TCL focal row, MCC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558197+00:00","notes":"Matched ERNIE-RNA jointly with TCL focal loss, then the MCC column. Table 2 explicitly reports test-set models. The cell is 0.657, distinct from AUROC 0.868."}}} {"id":"claim-b2-codonbert-vaccines-2024","kind":"claim","name":"Reported Spearman rho for CodonBERT","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["flu-vaccine mRNA property prediction"]},"source_ids":["codonbert-vaccines-2024"],"links":[{"relation":"subject","target_id":"b2-codonbert-vaccines-2024"}],"attributes":{"field":"attributes.printed_value","value":"0.81","source_locator":"Table 2, CodonBERT row, Flu vaccines Spearman correlation column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558201+00:00","notes":"Matched the CodonBERT row and Flu vaccines column (0.81). The table footnote identifies regression columns as Spearman rank correlation and singles out E. coli as classification; this is not a flu-vaccine accuracy score."}}} {"id":"claim-b2-dart-eval-regulatory-2024","kind":"claim","name":"Reported accuracy for DNABERT-2","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["regulatory element identification"]},"source_ids":["dart-eval-regulatory-2024"],"links":[{"relation":"subject","target_id":"b2-dart-eval-regulatory-2024"}],"attributes":{"field":"attributes.printed_value","value":"0.876","source_locator":"Table 3 (PDF page 5), DNABERT-2 row, Zero-Shot Accuracy column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558203+00:00","notes":"Inspected the pinned NeurIPS primary PDF table and explanatory text. The DNABERT-2 row reports 0.876 under Zero-Shot Accuracy. The caption defines this as pairwise prioritization of positives over matched controls, distinct from supervised absolute accuracy."}}} {"id":"claim-b2-dnabert2-enhancer-2025","kind":"claim","name":"Reported AUC for DNABERT2-Enhancer","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["enhancer recognition"]},"source_ids":["dnabert2-enhancer-2025"],"links":[{"relation":"subject","target_id":"b2-dnabert2-enhancer-2025"}],"attributes":{"field":"attributes.printed_value","value":"0.965","source_locator":"Table 4, first-layer DNABERT2-Enhancer row, AUC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558204+00:00","notes":"Resolved the first-layer row group. DNABERT2-Enhancer AUC is 0.965, whereas second-layer AUC is 0.933. The caption explicitly describes 5-fold cross-validation on Liu training data, not an independent held-out test."}}} {"id":"claim-b2-eden-genomic-classification-2026","kind":"claim","name":"Reported MCC for DNABERT-2","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["human core-promoter classification"]},"source_ids":["eden-genomic-classification-2026"],"links":[{"relation":"subject","target_id":"b2-eden-genomic-classification-2026"}],"attributes":{"field":"attributes.printed_value","value":"70.52","source_locator":"Table 5, DNABERT-2 row, H-CPD (MCC) column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:37.531Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-b2-ernie-rna-2025","kind":"claim","name":"Reported binary F1 for ERNIE-RNA","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary-structure prediction"]},"source_ids":["ernie-rna-2025"],"links":[{"relation":"subject","target_id":"b2-ernie-rna-2025"}],"attributes":{"field":"attributes.printed_value","value":"0.575","source_locator":"Table 2, ERNIE-RNA zero-shot row, bpRNA-new F1-Score (binary) column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558206+00:00","notes":"Resolved bpRNA-new as the first three-column dataset group and F1-Score (binary) as its third metric. ERNIE-RNA zero shot is 86M and reports 0.575; RNA3DB-2D F1 is instead 0.542."}}} {"id":"claim-b2-esm2-ofs-fitness-2025","kind":"claim","name":"Reported Spearman rho for ESM2 OFS pseudo-perplexity","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["protein variant fitness prediction"]},"source_ids":["esm2-ofs-fitness-2025"],"links":[{"relation":"subject","target_id":"b2-esm2-ofs-fitness-2025"}],"attributes":{"field":"attributes.printed_value","value":"0.403","source_locator":"Published PDF page 6 (033014-6), Table I, ESM2: OFS PP row, Aggregate mean column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source review, not human review","reviewed_at":"2026-09-16T10:45:41.099916+00:00","notes":"Publisher PDF retrieved through official APS harvest endpoint after direct download returned403. Table I is ProteinGym substitutions, not indels TableII. Last column aggregate mean0.403; separate function categories precede it. This verifies reported score, not experimental reproduction. Comparator rows in this table are sourced from ProteinGym; OFS PP is authors own method."}}} {"id":"claim-b2-fusion-breakpoint-foundation-models-2026","kind":"claim","name":"Reported ROC AUC for Nucleotide Transformer + NN (middle)","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["gene fusion breakpoint classification"]},"source_ids":["fusion-breakpoint-foundation-models-2026"],"links":[{"relation":"subject","target_id":"b2-fusion-breakpoint-foundation-models-2026"}],"attributes":{"field":"attributes.printed_value","value":"0.994","source_locator":"Table 2, NT / NN (middle) row, ROC AUC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558209+00:00","notes":"Matched NT jointly with NN (middle) and ROC AUC 0.994 in the full-test-set table. NT with SVM reports 0.995 and is a separate pipeline."}}} {"id":"claim-b2-genomic-tokenizer-selection-2025","kind":"claim","name":"Reported MCC for Caduceus (character tokens)","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["regulatory sequence classification"]},"source_ids":["genomic-tokenizer-selection-2025"],"links":[{"relation":"subject","target_id":"b2-genomic-tokenizer-selection-2025"}],"attributes":{"field":"attributes.printed_value","value":"0.778","source_locator":"Table 2, Regulatory row, Caduceus (char) MCC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558210+00:00","notes":"Matched Regulatory row with Caduceus (char) column, 0.778. Caption establishes these as MCC summaries by category; model-size row identifies 3.9M parameters. This is an aggregated category result, not a single unspecified split."}}} {"id":"claim-b2-gsmformer-ppi-2026","kind":"claim","name":"Reported AUROC for GSMFormer-PPI + ProstT5","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["protein-protein interaction prediction"]},"source_ids":["gsmformer-ppi-2026"],"links":[{"relation":"subject","target_id":"b2-gsmformer-ppi-2026"}],"attributes":{"field":"attributes.printed_value","value":"0.988","source_locator":"Table 6, ProstT5 embedding row, AUROC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558212+00:00","notes":"Matched ProstT5 embedding row and AUROC column, 0.988. Caption explicitly describes GSMFormer-PPI using embeddings as node features, not standalone ProstT5 prediction."}}} {"id":"claim-b2-megsite-2025","kind":"claim","name":"Reported AUC for MegSite + ESM3","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["DNA-binding residue prediction"]},"source_ids":["megsite-2025"],"links":[{"relation":"subject","target_id":"b2-megsite-2025"}],"attributes":{"field":"attributes.printed_value","value":"0.948","source_locator":"Table 2, DNA-129_Test / ESM3 row, AUC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558213+00:00","notes":"Resolved DNA-129_Test row group and ESM3 row. AUC is 0.948; the next numeric cell 0.582 is AP. Caption states an embedding comparison within MegSite."}}} {"id":"claim-b2-mrna-lm-2025","kind":"claim","name":"Reported Spearman rho for mRNA-LM","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["mRNA half-life prediction"]},"source_ids":["mrna-lm-2025"],"links":[{"relation":"subject","target_id":"b2-mrna-lm-2025"}],"attributes":{"field":"attributes.printed_value","value":"0.696","source_locator":"Table 1, mRNA-LM row, mRNA half-life Spearman column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558214+00:00","notes":"Resolved mRNA half-life column under the Spearman header spanning three tasks. mRNA-LM gives 0.696. Caption identifies average test performance across cross-validation splits; protein-expression AUROC is a different column."}}} {"id":"claim-b2-mrnabert-2025","kind":"claim","name":"Reported R-squared for mRNABERT","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["translation-efficiency prediction"]},"source_ids":["mrnabert-2025"],"links":[{"relation":"subject","target_id":"b2-mrnabert-2025"}],"attributes":{"field":"attributes.printed_value","value":"0.669","source_locator":"Table 2, mRNABERT (3066) row, Human R-squared column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558216+00:00","notes":"Resolved Human group and its R-squared subcolumn. mRNABERT (3066) reports 0.669; Human Spearman is 0.814 and Mouse R-squared is 0.649. Caption specifies ultra-long mRNA translation-efficiency prediction."}}} {"id":"claim-b2-mulan-2025","kind":"claim","name":"Reported AUC for MULAN-ESM2 S","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["human protein-protein interaction prediction"]},"source_ids":["mulan-2025"],"links":[{"relation":"subject","target_id":"b2-mulan-2025"}],"attributes":{"field":"attributes.printed_value","value":"0.717","source_locator":"Table 2, MULAN-ESM2 S row, HumanPPI AUC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558217+00:00","notes":"Resolved the multirow header: HumanPPI uses AUC. MULAN-ESM2 S has 0.717; this is the small-model group, distinct from M and L variants."}}} {"id":"claim-b2-phylogpn-2025","kind":"claim","name":"Reported AUROC for PhyloGPN","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["ClinVar 3-prime UTR variant classification"]},"source_ids":["phylogpn-2025"],"links":[{"relation":"subject","target_id":"b2-phylogpn-2025"}],"attributes":{"field":"attributes.printed_value","value":"0.94","source_locator":"Table 1, 3-prime UTR row, PhyloGPN AUROC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558218+00:00","notes":"Matched 3-prime UTR row and PhyloGPN column (0.94). Caption specifies log-likelihood-ratio predictions of ClinVar classes and explicitly defines each cell as AUROC."}}} {"id":"claim-b2-polya-glm-2025","kind":"claim","name":"Reported AUC for HyenaDNA","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["polyadenylation site detection"]},"source_ids":["polya-glm-2025"],"links":[{"relation":"subject","target_id":"b2-polya-glm-2025"}],"attributes":{"field":"attributes.printed_value","value":"0.7510","source_locator":"Table 1, Few-shot HyenaDNA row, G-G AUC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558220+00:00","notes":"Resolved Few-shot group, HyenaDNA row, and G-G subcolumn under AUC (0.7510). IG-G AUC is 0.7541. Caption states averages over five-fold cross-validation and distinguishes negative sampling regions."}}} {"id":"claim-b2-rlsite-rna-binding-2025","kind":"claim","name":"Reported AUC for RLsite","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA-small-molecule binding-site prediction"]},"source_ids":["rlsite-rna-binding-2025"],"links":[{"relation":"subject","target_id":"b2-rlsite-rna-binding-2025"}],"attributes":{"field":"attributes.printed_value","value":"0.828","source_locator":"Table 1, RLsite row, T18 AUC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558222+00:00","notes":"Matched RLsite and AUC (0.828). Caption explicitly identifies dataset T18; MCC 0.474 is a different metric."}}} {"id":"claim-b2-rnaret-2026","kind":"claim","name":"Reported F1 for RNAret","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["miRNA-mRNA interaction prediction"]},"source_ids":["rnaret-2026"],"links":[{"relation":"subject","target_id":"b2-rnaret-2026"}],"attributes":{"field":"attributes.printed_value","value":"0.9622","source_locator":"Table 1, MirTarRAW / 5-mer RNAret row, F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558224+00:00","notes":"Resolved the MirTarRAW section, 5-mer RNAret row, and F1 column (0.9622), distinct from DeepMirTarLeft F1 0.9728. Methods confirm 72/8/20 train/validation/test partition for MirTarRAW."}}} {"id":"claim-b2-spin-protein-function-2026","kind":"claim","name":"Reported F1 macro-weighted for SPIN + ESM2-35M","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["protein function annotation"]},"source_ids":["spin-protein-function-2026"],"links":[{"relation":"subject","target_id":"b2-spin-protein-function-2026"}],"attributes":{"field":"attributes.printed_value","value":"0.796","source_locator":"Table 1, ESM2-35M Test row, F1_m-w column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558225+00:00","notes":"Resolved Test group and macro-weighted F1 subcolumn (0.796) for frozen ESM2-35M in SPIN. Test weighted accuracy is 0.798. Methods define inverse-frequency class weighting for macro-weighted F1."}}} {"id":"claim-b2-structure-informed-plm-2025","kind":"claim","name":"Reported AUROC for structure-informed pLM","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["protein variant-effect classification"]},"source_ids":["structure-informed-plm-2025"],"links":[{"relation":"subject","target_id":"b2-structure-informed-plm-2025"}],"attributes":{"field":"attributes.printed_value","value":".803","source_locator":"PMC12068927 HTML, Table4, AA+SS+RSA+CM row, AUROC column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source review, not human review","reviewed_at":"2026-09-16T10:45:41.099916+00:00","notes":"Full-text HTML succeeds although EuropePMC XMLreturned404. Row is mutation-site variables AA+SS+RSA+CM, not neighbouring environment variant. AUROC .803 is numerically equivalent to preserved legacy0.803. Source check, not experimental reproduction; do not claim original source printed leading zero."}}} {"id":"claim-lit-001","kind":"claim","name":"Reported AUC for Caduceus-Ph","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["Human 5mC detection"]},"source_ids":["dna-foundation-models-2025"],"links":[{"relation":"subject","target_id":"lit-001"}],"attributes":{"field":"attributes.printed_value","value":"0.783","source_locator":"Table 3, Human 5mC row, Caduceus-Ph column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:35.327Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-002","kind":"claim","name":"Reported AUC for NT-v2","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["Human 5mC detection"]},"source_ids":["dna-foundation-models-2025"],"links":[{"relation":"subject","target_id":"lit-002"}],"attributes":{"field":"attributes.printed_value","value":"0.7377","source_locator":"Table 3, Human 5mC row, NT-v2 column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:35.327Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-003","kind":"claim","name":"Reported Accuracy for ENBED","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["Enhancer classification"]},"source_ids":["enbed-2024"],"links":[{"relation":"subject","target_id":"lit-003"}],"attributes":{"field":"attributes.printed_value","value":"90.3","source_locator":"Table 2, Mouse Enhancers row, ENBED column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:35.378Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-004","kind":"claim","name":"Reported Accuracy for ENBED (GRCh38)","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["Enhancer classification"]},"source_ids":["enbed-2024"],"links":[{"relation":"subject","target_id":"lit-004"}],"attributes":{"field":"attributes.printed_value","value":"81.1","source_locator":"Table 2, Mouse Enhancers row, ENBED (GRCh38) column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:35.378Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-005","kind":"claim","name":"Reported Accuracy for DNABERT-2","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["G-quadruplex classification"]},"source_ids":["quadruplex-llm-benchmark-2025"],"links":[{"relation":"subject","target_id":"lit-005"}],"attributes":{"field":"attributes.printed_value","value":"97.0","source_locator":"Table 5, DNABERT-2 (117 M) row, Accuracy column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:35.379Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-006","kind":"claim","name":"Reported Accuracy for Caduceus","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["G-quadruplex classification"]},"source_ids":["quadruplex-llm-benchmark-2025"],"links":[{"relation":"subject","target_id":"lit-006"}],"attributes":{"field":"attributes.printed_value","value":"95.0","source_locator":"Table 5, Caduceus (8 M) row, Accuracy column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:35.379Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-007","kind":"claim","name":"Reported AUROC for HyenaDNA","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["Enhancer-target gene prediction"]},"source_ids":["dnalongbench-2025"],"links":[{"relation":"subject","target_id":"lit-007"}],"attributes":{"field":"attributes.printed_value","value":"0.828","source_locator":"Table 3, HyenaDNA row, ETGP column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.492545+00:00","notes":"ETGP is the first numeric column, separate from the six CMP cell-type columns and average; caption defines ETGP AUROC. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-008","kind":"claim","name":"Reported AUROC for Caduceus-Ph","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["Enhancer-target gene prediction"]},"source_ids":["dnalongbench-2025"],"links":[{"relation":"subject","target_id":"lit-008"}],"attributes":{"field":"attributes.printed_value","value":"0.826","source_locator":"Table 3, Caduceus-Ph row, ETGP column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.493765+00:00","notes":"ETGP is the first numeric column, separate from the six CMP cell-type columns and average; caption defines ETGP AUROC. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-009","kind":"claim","name":"Reported Pearson R for RiNALMo","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["Mean ribosome load from MPRA"]},"source_ids":["mrnabench-2025"],"links":[{"relation":"subject","target_id":"lit-009"}],"attributes":{"field":"attributes.printed_value","value":"0.74","source_locator":"Table 2, RiNALMo row, MRL MPRA column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.497221+00:00","notes":"MRL MPRA is the second task under Local and uses R. Caption specifies mean over ten random seeds and selected best model per family, not a fully identified checkpoint. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-010","kind":"claim","name":"Reported Pearson R for RNA-FM","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["Mean ribosome load from MPRA"]},"source_ids":["mrnabench-2025"],"links":[{"relation":"subject","target_id":"lit-010"}],"attributes":{"field":"attributes.printed_value","value":"0.49","source_locator":"Table 2, RNA-FM row, MRL MPRA column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.500211+00:00","notes":"MRL MPRA is the second task under Local and uses R. Caption specifies mean over ten random seeds and selected best model per family, not a fully identified checkpoint. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-011","kind":"claim","name":"Reported F1 for BPfold","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary structure"]},"source_ids":["bpfold-2025"],"links":[{"relation":"subject","target_id":"lit-011"}],"attributes":{"field":"attributes.printed_value","value":"0.814","source_locator":"Table 2, BPfold row, PDB F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.502000+00:00","notes":"PDB is the second four-metric block; its F1 is numeric column six, not Rfam F1. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-012","kind":"claim","name":"Reported F1 for RNAfold","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary structure"]},"source_ids":["bpfold-2025"],"links":[{"relation":"subject","target_id":"lit-012"}],"attributes":{"field":"attributes.printed_value","value":"0.747","source_locator":"Table 2, RNAfold row, PDB F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.504220+00:00","notes":"PDB is the second four-metric block; its F1 is numeric column six, not Rfam F1. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-013","kind":"claim","name":"Reported F1 for TU-Fold (aug)","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary structure"]},"source_ids":["tu-fold-2025"],"links":[{"relation":"subject","target_id":"lit-013"}],"attributes":{"field":"attributes.printed_value","value":"0.947","source_locator":"Table 2, TU-Fold (aug) row, Overall F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.505799+00:00","notes":"Overall is the first two-metric block. F1 is first numeric column; source cell includes uncertainty after the preserved central value. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-014","kind":"claim","name":"Reported F1 for UFold","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary structure"]},"source_ids":["tu-fold-2025"],"links":[{"relation":"subject","target_id":"lit-014"}],"attributes":{"field":"attributes.printed_value","value":"0.938","source_locator":"Table 2, UFold row, Overall F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.507031+00:00","notes":"Overall is the first two-metric block. F1 is first numeric column; source cell includes uncertainty after the preserved central value. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-015","kind":"claim","name":"Reported Median F1 for DEBFold","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary structure"]},"source_ids":["debfold-2024"],"links":[{"relation":"subject","target_id":"lit-015"}],"attributes":{"field":"attributes.printed_value","value":"55.7","source_locator":"Table 1, DEBFold row, TestSetβ F1 (%) column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.509290+00:00","notes":"TestSet beta is the second four-column block. F1 (%) is its first column; caption reports test-set median F1. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-016","kind":"claim","name":"Reported Median F1 for RNAfold","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary structure"]},"source_ids":["debfold-2024"],"links":[{"relation":"subject","target_id":"lit-016"}],"attributes":{"field":"attributes.printed_value","value":"52.3","source_locator":"Table 1, RNAfold row, TestSetβ F1 (%) column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.511166+00:00","notes":"TestSet beta is the second four-column block. F1 (%) is its first column; caption reports test-set median F1. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-017","kind":"claim","name":"Reported Mean Spearman rho for ESM-2","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["Zero-shot substitution mutation effects: stability"]},"source_ids":["proteingym-2023"],"links":[{"relation":"subject","target_id":"lit-017"}],"attributes":{"field":"attributes.printed_value","value":"0.488","source_locator":"Table A7, ESM-2 (15B) row, Stability column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.517323+00:00","notes":"Table A7 is zero-shot substitution DMS grouped by function. Stability is the fifth numeric column; model-type row spans do not change its placement. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-018","kind":"claim","name":"Reported Mean Spearman rho for ProteinMPNN","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["Zero-shot substitution mutation effects: stability"]},"source_ids":["proteingym-2023"],"links":[{"relation":"subject","target_id":"lit-018"}],"attributes":{"field":"attributes.printed_value","value":"0.566","source_locator":"Table A7, ProteinMPNN row, Stability column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.523422+00:00","notes":"Table A7 is zero-shot substitution DMS grouped by function. Stability is the fifth numeric column; model-type row spans do not change its placement. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-019","kind":"claim","name":"Reported AUROC for FUJISAN","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["Enzyme functional identity prediction"]},"source_ids":["fujisan-2024"],"links":[{"relation":"subject","target_id":"lit-019"}],"attributes":{"field":"attributes.printed_value","value":"0.9427","source_locator":"Table 1, FUJISAN row, AUROC column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:35.728Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-020","kind":"claim","name":"Reported AUROC for ESM2","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["Enzyme functional identity prediction"]},"source_ids":["fujisan-2024"],"links":[{"relation":"subject","target_id":"lit-020"}],"attributes":{"field":"attributes.printed_value","value":"0.7991","source_locator":"Table 1, ESM2 row, AUROC column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:35.728Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-021","kind":"claim","name":"Reported R² for ESM-2","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["Mutated RBD binding prediction"]},"source_ids":["prime-2026"],"links":[{"relation":"subject","target_id":"lit-021"}],"attributes":{"field":"attributes.printed_value","value":"0.0248","source_locator":"Table 1, ESM-2 8M / Mean / not fine-tuned row, Position-Stratified Split Binding R² column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.525183+00:00","notes":"Resolved model row spans and Mean/CLS subrows in JATS: selected Mean, not fine-tuned (cross), Position-Stratified Split > Binding > R-squared. Central value agrees; uncertainty is retained in evidence. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-022","kind":"claim","name":"Reported R² for ESM-C","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["Mutated RBD binding prediction"]},"source_ids":["prime-2026"],"links":[{"relation":"subject","target_id":"lit-022"}],"attributes":{"field":"attributes.printed_value","value":"-0.0162","source_locator":"Table 1, ESM-C 300M / Mean / not fine-tuned row, Position-Stratified Split Binding R² column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.526387+00:00","notes":"Resolved model row spans and Mean/CLS subrows in JATS: selected Mean, not fine-tuned (cross), Position-Stratified Split > Binding > R-squared. Central value agrees; uncertainty is retained in evidence. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-023","kind":"claim","name":"Reported Mean |Spearman rho| for PST","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["Zero-shot variant effect prediction"]},"source_ids":["pst-2025"],"links":[{"relation":"subject","target_id":"lit-023"}],"attributes":{"field":"attributes.printed_value","value":"0.501","source_locator":"Table 2, PST row, Zero-shot VEP Mean |ρ| column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.527633+00:00","notes":"Zero-shot VEP is the last metric group; selected Mean absolute rho, not GO/EC/binding-site metrics. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-024","kind":"claim","name":"Reported Mean |Spearman rho| for ESM-2","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["Zero-shot variant effect prediction"]},"source_ids":["pst-2025"],"links":[{"relation":"subject","target_id":"lit-024"}],"attributes":{"field":"attributes.printed_value","value":"0.489","source_locator":"Table 2, ESM-2 row, Zero-shot VEP Mean |ρ| column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.528656+00:00","notes":"Zero-shot VEP is the last metric group; selected Mean absolute rho, not GO/EC/binding-site metrics. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-025","kind":"claim","name":"Reported F1-Score for scGPT","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Cell-type identification"]},"source_ids":["single-cell-peft-2024"],"links":[{"relation":"subject","target_id":"lit-025"}],"attributes":{"field":"attributes.printed_value","value":"0.734","source_locator":"Table 2, M.S. / scGPT row, F1-Score column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.530269+00:00","notes":"Selected M.S. dataset block, first scGPT/Geneformer occurrences. F1-Score is last column; later dataset blocks deliberately excluded. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-026","kind":"claim","name":"Reported F1-Score for Geneformer","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Cell-type identification"]},"source_ids":["single-cell-peft-2024"],"links":[{"relation":"subject","target_id":"lit-026"}],"attributes":{"field":"attributes.printed_value","value":"0.388","source_locator":"Table 2, M.S. / Geneformer row, F1-Score column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.531756+00:00","notes":"Selected M.S. dataset block, first scGPT/Geneformer occurrences. F1-Score is last column; later dataset blocks deliberately excluded. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-027","kind":"claim","name":"Reported Partial-label accuracy for C2S (GPT-2 Large)","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Combinatorial cell-label classification"]},"source_ids":["cell2sentence-2024"],"links":[{"relation":"subject","target_id":"lit-027"}],"attributes":{"field":"attributes.printed_value","value":"0.631","source_locator":"Table 3, Partial label / C2S (GPT-2 Large) row, L1000 Acc column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.533640+00:00","notes":"Read inline small-caps/bold XML in document order, restoring Geneformer and GPT-2 Large labels. Selected Partial label (first block), L1000 > Acc, not AUROC or Full label. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-028","kind":"claim","name":"Reported Partial-label accuracy for Geneformer","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Combinatorial cell-label classification"]},"source_ids":["cell2sentence-2024"],"links":[{"relation":"subject","target_id":"lit-028"}],"attributes":{"field":"attributes.printed_value","value":"0.419","source_locator":"Table 3, Partial label / Geneformer row, L1000 Acc column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.535220+00:00","notes":"Read inline small-caps/bold XML in document order, restoring Geneformer and GPT-2 Large labels. Selected Partial label (first block), L1000 > Acc, not AUROC or Full label. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-029","kind":"claim","name":"Reported F1 for scGPT","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Cell-type annotation"]},"source_ids":["scelmo-2025"],"links":[{"relation":"subject","target_id":"lit-029"}],"attributes":{"field":"attributes.printed_value","value":"0.550","source_locator":"Table 1, hPancreas zero-shot / scGPT (z) row, F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.537541+00:00","notes":"Selected first hPancreas zero-shot block and F1 last column. Caption says some scores are copied from GenePT; this is source checking of the reported table, not independent experimental evidence. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-030","kind":"claim","name":"Reported F1 for Geneformer","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Cell-type annotation"]},"source_ids":["scelmo-2025"],"links":[{"relation":"subject","target_id":"lit-030"}],"attributes":{"field":"attributes.printed_value","value":"0.270","source_locator":"Table 1, hPancreas zero-shot / Geneformer (z) row, F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.539630+00:00","notes":"Selected first hPancreas zero-shot block and F1 last column. Caption says some scores are copied from GenePT; this is source checking of the reported table, not independent experimental evidence. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-031","kind":"claim","name":"Reported AUROC for scRegNet (Geneformer backbone)","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Gene-regulatory link prediction"]},"source_ids":["scregnet-2025"],"links":[{"relation":"subject","target_id":"lit-031"}],"attributes":{"field":"attributes.printed_value","value":"0.89","source_locator":"Table 2, scRegNet (w/ Geneformer) row, hESC AUROC entry","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.541287+00:00","notes":"Read break elements: cells contain AUROC on first line then AUPRC. Selected hESC (first cell type), first line. Caption specifies 500 most-variable genes and 50 independent evaluations. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-032","kind":"claim","name":"Reported AUROC for scRegNet (scBERT backbone)","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Gene-regulatory link prediction"]},"source_ids":["scregnet-2025"],"links":[{"relation":"subject","target_id":"lit-032"}],"attributes":{"field":"attributes.printed_value","value":"0.88","source_locator":"Table 2, scRegNet (w/ scBERT) row, hESC AUROC entry","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.542807+00:00","notes":"Read break elements: cells contain AUROC on first line then AUPRC. Selected hESC (first cell type), first line. Caption specifies 500 most-variable genes and 50 independent evaluations. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-033","kind":"claim","name":"Reported Accuracy for ProkBERT-mini","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["E. coli sigma70 promoter prediction"]},"source_ids":["prokbert-2024"],"links":[{"relation":"subject","target_id":"lit-033"}],"attributes":{"field":"attributes.printed_value","value":"0.87","source_locator":"Table 3, ProkBERT-mini row, Accuracy column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:36.197Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-034","kind":"claim","name":"Reported Accuracy for Promotech","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["E. coli sigma70 promoter prediction"]},"source_ids":["prokbert-2024"],"links":[{"relation":"subject","target_id":"lit-034"}],"attributes":{"field":"attributes.printed_value","value":"0.71","source_locator":"Table 3, Promotech row, Accuracy column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:36.197Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-035","kind":"claim","name":"Reported Promoter-class F1 for Eco70PromBERT","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["E. coli sigma70 promoter prediction"]},"source_ids":["cyaprombert-2022"],"links":[{"relation":"subject","target_id":"lit-035"}],"attributes":{"field":"attributes.printed_value","value":"0.91","source_locator":"TABLE 3, Eco70PromBERT (BERT-base + 1bp tokenizer) row, F1 score Promoter column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.544033+00:00","notes":"F1 score is the third two-column group; selected Promoter subcolumn, not AUROC or precision. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-036","kind":"claim","name":"Reported Promoter-class F1 for iPro70-FMWin","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["E. coli sigma70 promoter prediction"]},"source_ids":["cyaprombert-2022"],"links":[{"relation":"subject","target_id":"lit-036"}],"attributes":{"field":"attributes.printed_value","value":"0.90","source_locator":"TABLE 3, iPro70-FMWin row, F1 score Promoter column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.544926+00:00","notes":"F1 score is the third two-column group; selected Promoter subcolumn, not AUROC or precision. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-037","kind":"claim","name":"Reported MCC for EVO2","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Genome-wide prophage detection"]},"source_ids":["lambda-prophage-2026"],"links":[{"relation":"subject","target_id":"lit-037"}],"attributes":{"field":"attributes.printed_value","value":"0.680","source_locator":"Table 5, EVO2 row, MCC column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:36.240Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-038","kind":"claim","name":"Reported MCC for geNomad","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Genome-wide prophage detection"]},"source_ids":["lambda-prophage-2026"],"links":[{"relation":"subject","target_id":"lit-038"}],"attributes":{"field":"attributes.printed_value","value":"0.794","source_locator":"Table 5, geNomad row, MCC column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:36.240Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-039","kind":"claim","name":"Reported F1 score for NABAS+","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Metagenomic taxonomic classification"]},"source_ids":["nabas-plus-2025"],"links":[{"relation":"subject","target_id":"lit-039"}],"attributes":{"field":"attributes.printed_value","value":"0.719","source_locator":"Table 3, Sample19-new / NABAS+ row, F1 score column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.546107+00:00","notes":"Selected Sample19-new explicitly, not Sample19-old, and F1 rather than precision/recall. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-040","kind":"claim","name":"Reported F1 score for MetaPhlAn3","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Metagenomic taxonomic classification"]},"source_ids":["nabas-plus-2025"],"links":[{"relation":"subject","target_id":"lit-040"}],"attributes":{"field":"attributes.printed_value","value":"0.753","source_locator":"Table 3, Sample19-new / MetaPhlAn3 row, F1 score column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.547023+00:00","notes":"Selected Sample19-new explicitly, not Sample19-old, and F1 rather than precision/recall. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-041","kind":"claim","name":"Reported Success rate, ligand all-atom RMSD <2 Å for Chai-1","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Lipid–protein binding pose"]},"source_ids":["lipp-2026"],"links":[{"relation":"subject","target_id":"lit-041"}],"attributes":{"field":"attributes.printed_value","value":"60.7","source_locator":"Table 2, Chai-1 row, LiPP (N=331) % Success Rate column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.548973+00:00","notes":"JATS label is bare 2, which caused original parser miss. LiPP N=331 full-set column selected, not N=36 test subset. Caption success is lipid all-atom RMSD <2 Angstrom; PB-valid is a separate table. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-042","kind":"claim","name":"Reported Success rate, ligand all-atom RMSD <2 Å for DiffDock-L","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Lipid–protein binding pose"]},"source_ids":["lipp-2026"],"links":[{"relation":"subject","target_id":"lit-042"}],"attributes":{"field":"attributes.printed_value","value":"46.8","source_locator":"Table 2, DiffDock-L row, LiPP (N=331) % Success Rate column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.550691+00:00","notes":"JATS label is bare 2, which caused original parser miss. LiPP N=331 full-set column selected, not N=36 test subset. Caption success is lipid all-atom RMSD <2 Angstrom; PB-valid is a separate table. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-043","kind":"claim","name":"Reported Forward-screening success rate for DiffDock-NMDN","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand virtual screening"]},"source_ids":["nmdn-2025"],"links":[{"relation":"subject","target_id":"lit-043"}],"attributes":{"field":"attributes.printed_value","value":"66.7","source_locator":"Table 2, DiffDock-NMDN / NMDN row, success rate (%) column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.552697+00:00","notes":"All scoring functions share DiffDock-NMDN poses via rowspan. Selected forward-screening success percentage, not docking pose success or scoring-power correlation. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-044","kind":"claim","name":"Reported Forward-screening success rate for Vina","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand virtual screening"]},"source_ids":["nmdn-2025"],"links":[{"relation":"subject","target_id":"lit-044"}],"attributes":{"field":"attributes.printed_value","value":"42.1","source_locator":"Table 2, Vina scoring row, success rate (%) column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.554518+00:00","notes":"All scoring functions share DiffDock-NMDN poses via rowspan. Selected forward-screening success percentage, not docking pose success or scoring-power correlation. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-045","kind":"claim","name":"Reported Median ligand RMSD for Boltz-1","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand pose prediction"]},"source_ids":["boltz-stereochemistry-2025"],"links":[{"relation":"subject","target_id":"lit-045"}],"attributes":{"field":"attributes.printed_value","value":"1.393","source_locator":"Table 1, Boltz-1 row, Ligand RMSD (Å) column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.555674+00:00","notes":"JATS label is bare 1. Selected Ligand RMSD column in Plinder-L95, not Protein RMSD; retained method row without refinement settings. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-046","kind":"claim","name":"Reported Median ligand RMSD for DiffDock","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand pose prediction"]},"source_ids":["boltz-stereochemistry-2025"],"links":[{"relation":"subject","target_id":"lit-046"}],"attributes":{"field":"attributes.printed_value","value":"1.342","source_locator":"Table 1, DiffDock row, Ligand RMSD (Å) column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.556572+00:00","notes":"JATS label is bare 1. Selected Ligand RMSD column in Plinder-L95, not Protein RMSD; retained method row without refinement settings. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-047","kind":"claim","name":"Reported Pearson R for Boltz-2","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Ligand potency prediction using generated poses"]},"source_ids":["mpro-pose-affinity-2025"],"links":[{"relation":"subject","target_id":"lit-047"}],"attributes":{"field":"attributes.printed_value","value":"0.800","source_locator":"Table 3, Boltz-2 row, Pearson’s R column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.557756+00:00","notes":"JATS label is bare 3. Selected SARS-CoV-2 Mpro potency Pearson R, not MERS-CoV table 2 or Boltz-2-Internal row; uncertainty retained in evidence. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-048","kind":"claim","name":"Reported Pearson R for DiffDock","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Ligand potency prediction using generated poses"]},"source_ids":["mpro-pose-affinity-2025"],"links":[{"relation":"subject","target_id":"lit-048"}],"attributes":{"field":"attributes.printed_value","value":"0.695","source_locator":"Table 3, DiffDock row, Pearson’s R column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.558815+00:00","notes":"JATS label is bare 3. Selected SARS-CoV-2 Mpro potency Pearson R, not MERS-CoV table 2 or Boltz-2-Internal row; uncertainty retained in evidence. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-b3-003","kind":"claim","name":"Reported F1 for Mouse-Geneformer","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Human thymus cell-type classification"]},"source_ids":["mouse-geneformer-2025"],"links":[{"relation":"subject","target_id":"lit-b3-003"}],"attributes":{"field":"attributes.printed_value","value":"48.57","source_locator":"Table 4, h/ Thymus row, Mouse-Geneformer Zero-shot F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.392488+00:00","notes":"h/ Thymus row, four human cell types; zero-shot model blocks use Acc then F1, not fine-tuned scores. Ortholog-based conversion evaluated, not mouse cell annotation. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-004","kind":"claim","name":"Reported F1 for Human-Geneformer","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Human thymus cell-type classification"]},"source_ids":["mouse-geneformer-2025"],"links":[{"relation":"subject","target_id":"lit-b3-004"}],"attributes":{"field":"attributes.printed_value","value":"74.48","source_locator":"Table 4, h/ Thymus row, Human-Geneformer Zero-shot F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.394073+00:00","notes":"h/ Thymus row, four human cell types; zero-shot model blocks use Acc then F1, not fine-tuned scores. Ortholog-based conversion evaluated, not mouse cell annotation. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-005","kind":"claim","name":"Reported F1 for scLLMDA","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Cross-platform scATAC cell-type annotation"]},"source_ids":["scatac-llmda-2026"],"links":[{"relation":"subject","target_id":"lit-b3-005"}],"attributes":{"field":"attributes.printed_value","value":"0.6525","source_locator":"Table 2, scLLMDA row, Ref: MosA1 / Q: WholeBrainA F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.395850+00:00","notes":"First reference/query block MosA1 to WholeBrainA, F1 second column in block; direction of transfer is part of protocol. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-006","kind":"claim","name":"Reported F1 for MINGLE","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Cross-platform scATAC cell-type annotation"]},"source_ids":["scatac-llmda-2026"],"links":[{"relation":"subject","target_id":"lit-b3-006"}],"attributes":{"field":"attributes.printed_value","value":"0.6256","source_locator":"Table 2, MINGLE row, Ref: MosA1 / Q: WholeBrainA F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.397272+00:00","notes":"First reference/query block MosA1 to WholeBrainA, F1 second column in block; direction of transfer is part of protocol. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-011","kind":"claim","name":"Reported Adjusted Rand Index for GenePT-w","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Cell-type structure in frozen embeddings"]},"source_ids":["genept-2024"],"links":[{"relation":"subject","target_id":"lit-b3-011"}],"attributes":{"field":"attributes.printed_value","value":"0.54","source_locator":"Table 2, Aorta / Cell type row, GenePT-w ARI column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.399274+00:00","notes":"First Cell type row belongs to Aorta, not preceding Phenotype row or later organs. ARI is first in each three-metric method block, not AMI/ASW. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-012","kind":"claim","name":"Reported Adjusted Rand Index for scGPT","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Cell-type structure in frozen embeddings"]},"source_ids":["genept-2024"],"links":[{"relation":"subject","target_id":"lit-b3-012"}],"attributes":{"field":"attributes.printed_value","value":"0.47","source_locator":"Table 2, Aorta / Cell type row, scGPT ARI column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.402215+00:00","notes":"First Cell type row belongs to Aorta, not preceding Phenotype row or later organs. ARI is first in each three-metric method block, not AMI/ASW. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-013","kind":"claim","name":"Reported Balanced accuracy for Best frozen single-cell foundation model","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Donor-aware age-class prediction"]},"source_ids":["single-cell-aging-probes-2026"],"links":[{"relation":"subject","target_id":"lit-b3-013"}],"attributes":{"field":"attributes.printed_value","value":"0.322","source_locator":"Table 2, AIDA v2 row, scFM BA ± SD column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:44.421Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b3-014","kind":"claim","name":"Reported Balanced accuracy for Gene-expression PCA","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Donor-aware age-class prediction"]},"source_ids":["single-cell-aging-probes-2026"],"links":[{"relation":"subject","target_id":"lit-b3-014"}],"attributes":{"field":"attributes.printed_value","value":"0.384","source_locator":"Table 2, AIDA v2 row, Gene-expr BA column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:44.421Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b3-015","kind":"claim","name":"Reported Cell-type accuracy for scaLR","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["PBMC cell-type classification"]},"source_ids":["scalr-2025"],"links":[{"relation":"subject","target_id":"lit-b3-015"}],"attributes":{"field":"attributes.printed_value","value":"0.942","source_locator":"Table 2, scaLR row, Cell type Accuracy column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.403844+00:00","notes":"PBMCs-BS all-feature/all-sample cell-type accuracy block, not cell-state accuracy or time. Footnote letters on model labels excluded from identity. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-016","kind":"claim","name":"Reported Cell-type accuracy for scVI + scANVI","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["PBMC cell-type classification"]},"source_ids":["scalr-2025"],"links":[{"relation":"subject","target_id":"lit-b3-016"}],"attributes":{"field":"attributes.printed_value","value":"0.939","source_locator":"Table 2, Svi-tools (scVI & scANVI) row, Cell type Accuracy column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.405089+00:00","notes":"PBMCs-BS all-feature/all-sample cell-type accuracy block, not cell-state accuracy or time. Footnote letters on model labels excluded from identity. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-017","kind":"claim","name":"Reported AUC for scXDR","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Cross-dataset single-cell drug response transfer"]},"source_ids":["scxdr-2026"],"links":[{"relation":"subject","target_id":"lit-b3-017"}],"attributes":{"field":"attributes.printed_value","value":"0.8248","source_locator":"Table 2, scXDR row, Scenario 2 column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:50.056Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b3-018","kind":"claim","name":"Reported AUC for scVI","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Cross-dataset single-cell drug response transfer"]},"source_ids":["scxdr-2026"],"links":[{"relation":"subject","target_id":"lit-b3-018"}],"attributes":{"field":"attributes.printed_value","value":"0.6970","source_locator":"Table 2, scVI row, Scenario 2 column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:50.056Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b3-019","kind":"claim","name":"Reported L1 abundance error for CAMMiQ","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Strain-level abundance quantification"]},"source_ids":["cammiq-2022"],"links":[{"relation":"subject","target_id":"lit-b3-019"}],"attributes":{"field":"attributes.printed_value","value":"0.0517","source_locator":"Table 5, B. L1 Err. / HumanGut-all row, CAMMiQ L1 Err. column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.408237+00:00","notes":"HumanGut-all in B. L1 Err. block (second occurrence), not strain count or C. L2 Err. Numbers are errors; lower is better. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-020","kind":"claim","name":"Reported L1 abundance error for Kraken2","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Strain-level abundance quantification"]},"source_ids":["cammiq-2022"],"links":[{"relation":"subject","target_id":"lit-b3-020"}],"attributes":{"field":"attributes.printed_value","value":"0.2841","source_locator":"Table 5, B. L1 Err. / HumanGut-all row, Kraken2 L1 Err. column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.411180+00:00","notes":"HumanGut-all in B. L1 Err. block (second occurrence), not strain count or C. L2 Err. Numbers are errors; lower is better. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-021","kind":"claim","name":"Reported Genus-level F1 for Lazypipe-nt","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Simulated metagenome virus-taxon retrieval"]},"source_ids":["lazypipe-2020"],"links":[{"relation":"subject","target_id":"lit-b3-021"}],"attributes":{"field":"attributes.printed_value","value":"0.932","source_locator":"Table 1, Lazypipe-nt / Genus row, F column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.412605+00:00","notes":"First Genus block selected using rank row span, not Species. Final F column is F score, not precision or recall. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-022","kind":"claim","name":"Reported Genus-level F1 for Kraken2","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Simulated metagenome virus-taxon retrieval"]},"source_ids":["lazypipe-2020"],"links":[{"relation":"subject","target_id":"lit-b3-022"}],"attributes":{"field":"attributes.printed_value","value":"0.627","source_locator":"Table 1, Kraken2 / Genus row, F column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.413635+00:00","notes":"First Genus block selected using rank row span, not Species. Final F column is F score, not precision or recall. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-023","kind":"claim","name":"Reported Macro F1 for NCD-gzip","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["CAMI II superkingdom read classification"]},"source_ids":["ncd-metagenomics-2026"],"links":[{"relation":"subject","target_id":"lit-b3-023"}],"attributes":{"field":"attributes.printed_value","value":"0.9804","source_locator":"Table 5, NCD Superkingdom row, F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.414806+00:00","notes":"NCD rank-specific table 5, F1 column; Superkingdom and Phylum are different classification granularities. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-024","kind":"claim","name":"Reported Macro F1 for NCD-gzip","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["CAMI II phylum read classification"]},"source_ids":["ncd-metagenomics-2026"],"links":[{"relation":"subject","target_id":"lit-b3-024"}],"attributes":{"field":"attributes.printed_value","value":"0.1263","source_locator":"Table 5, NCD Phylum row, F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.415788+00:00","notes":"NCD rank-specific table 5, F1 column; Superkingdom and Phylum are different classification granularities. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-025","kind":"claim","name":"Reported Average prophage F1 for VIBRANT","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Simulated prophage-contig detection"]},"source_ids":["viral-contig-simulation-2021"],"links":[{"relation":"subject","target_id":"lit-b3-025"}],"attributes":{"field":"attributes.printed_value","value":"0.169","source_locator":"Table 3, Vibrant row, Prophage F1 column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:50.134Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b3-026","kind":"claim","name":"Reported Average prophage F1 for VirSorter","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Simulated prophage-contig detection"]},"source_ids":["viral-contig-simulation-2021"],"links":[{"relation":"subject","target_id":"lit-b3-026"}],"attributes":{"field":"attributes.printed_value","value":"0.147","source_locator":"Table 3, VirSorter row, Prophage F1 column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:50.134Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b3-027","kind":"claim","name":"Reported F1 for GenomeOcean","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Natural vs artificial microbial genome sequence"]},"source_ids":["genomeocean-2025"],"links":[{"relation":"subject","target_id":"lit-b3-027"}],"attributes":{"field":"attributes.printed_value","value":"99.03","source_locator":"Table 2, GenomeOcean row, F1 column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:55.224Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b3-028","kind":"claim","name":"Reported F1 for DNABERT-2","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Natural vs artificial microbial genome sequence"]},"source_ids":["genomeocean-2025"],"links":[{"relation":"subject","target_id":"lit-b3-028"}],"attributes":{"field":"attributes.printed_value","value":"85.12","source_locator":"Table 2, DNABERT-2 row, F1 column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:55.224Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b3-029","kind":"claim","name":"Reported Genus-level F1 for kMetaShot","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Mock-community MAG taxonomy classification"]},"source_ids":["kmetashot-2025"],"links":[{"relation":"subject","target_id":"lit-b3-029"}],"attributes":{"field":"attributes.printed_value","value":"95.83","source_locator":"Table 2, F1-score % row, Genus kMS column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.417367+00:00","notes":"Real mock sequencing table, F1-score percentage row; Genus is final three-column block, selecting kMS or Gtk rather than Species/Strain. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-030","kind":"claim","name":"Reported Genus-level F1 for GTDB-Tk","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Mock-community MAG taxonomy classification"]},"source_ids":["kmetashot-2025"],"links":[{"relation":"subject","target_id":"lit-b3-030"}],"attributes":{"field":"attributes.printed_value","value":"89.80","source_locator":"Table 2, F1-score % row, Genus Gtk column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.418914+00:00","notes":"Real mock sequencing table, F1-score percentage row; Genus is final three-column block, selecting kMS or Gtk rather than Species/Strain. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-031","kind":"claim","name":"Reported F1 for Lemur","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Long-read taxonomic profiling"]},"source_ids":["lemur-magnet-2024"],"links":[{"relation":"subject","target_id":"lit-b3-031"}],"attributes":{"field":"attributes.printed_value","value":"0.376","source_locator":"Table 3, LOG 10% / Lemur row, F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.420493+00:00","notes":"LOG 10% first block, F1 column. Kraken 2 row inherits dataset via rowspan; not LOG 75% or abundance Spearman. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-032","kind":"claim","name":"Reported F1 for Kraken 2","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Long-read taxonomic profiling"]},"source_ids":["lemur-magnet-2024"],"links":[{"relation":"subject","target_id":"lit-b3-032"}],"attributes":{"field":"attributes.printed_value","value":"0.375","source_locator":"Table 3, LOG 10% / Kraken 2 row, F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.421888+00:00","notes":"LOG 10% first block, F1 column. Kraken 2 row inherits dataset via rowspan; not LOG 75% or abundance Spearman. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-033","kind":"claim","name":"Reported Mean AUC for iPro-MP","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Multi-species prokaryotic promoter detection"]},"source_ids":["ipromp-2025"],"links":[{"relation":"subject","target_id":"lit-b3-033"}],"attributes":{"field":"attributes.printed_value","value":"0.935","source_locator":"Table 2, iPro-MP row, AUC column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:55.361Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b3-034","kind":"claim","name":"Reported Mean AUC for Prompt","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Multi-species prokaryotic promoter detection"]},"source_ids":["ipromp-2025"],"links":[{"relation":"subject","target_id":"lit-b3-034"}],"attributes":{"field":"attributes.printed_value","value":"0.835","source_locator":"Table 2, Prompt row, AUC column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:55.361Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b3-035","kind":"claim","name":"Reported Genus macro AveP for ICCTax","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Hierarchical metagenomic taxonomy classification"]},"source_ids":["icctax-2025"],"links":[{"relation":"subject","target_id":"lit-b3-035"}],"attributes":{"field":"attributes.printed_value","value":"67.20","source_locator":"Table 2, ICCTax row, Genus column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:55.373Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b3-036","kind":"claim","name":"Reported Genus macro AveP for Kraken2","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Hierarchical metagenomic taxonomy classification"]},"source_ids":["icctax-2025"],"links":[{"relation":"subject","target_id":"lit-b3-036"}],"attributes":{"field":"attributes.printed_value","value":"70.56","source_locator":"Table 2, Kraken2 row, Genus column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:55.373Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b3-037","kind":"claim","name":"Reported AUC-ROC for Chai-1","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Antibody–antigen interaction prediction using folded complexes"]},"source_ids":["antibody-flexibility-2025"],"links":[{"relation":"subject","target_id":"lit-b3-037"}],"attributes":{"field":"attributes.printed_value","value":"0.86","source_locator":"Table 5, Folded row, Chai-1 (no MSA) column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:55.400Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b3-038","kind":"claim","name":"Reported AUC-ROC for Boltz-1","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Antibody–antigen interaction prediction using folded complexes"]},"source_ids":["antibody-flexibility-2025"],"links":[{"relation":"subject","target_id":"lit-b3-038"}],"attributes":{"field":"attributes.printed_value","value":"0.85","source_locator":"Table 5, Folded row, Boltz-1 (no MSA) column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:55.400Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b3-039","kind":"claim","name":"Reported Top-1 ligand RMSD <2 Å rate for Boltz-1","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand pose prediction"]},"source_ids":["boltz1-2025"],"links":[{"relation":"subject","target_id":"lit-b3-039"}],"attributes":{"field":"attributes.printed_value","value":"0.545","source_locator":"Table 1, 3 recycling rounds / 200 steps row, L-RMSD <2Å top-1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.424864+00:00","notes":"3 recycling rounds and 200 steps; L-RMSD <2 Angstrom top-1 (last column), not oracle. Five samples generated; top-1 means highest-confidence candidate. Repeated reference rows are one evaluation, not independent experiments. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-040","kind":"claim","name":"Reported Mean CDR H3 RMSD for Ibex","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Antibody loop structure prediction"]},"source_ids":["ibex-2025"],"links":[{"relation":"subject","target_id":"lit-b3-040"}],"attributes":{"field":"attributes.printed_value","value":"2.72","source_locator":"Table 1, Antibodies / Ibex row, CDR H3 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.426811+00:00","notes":"First Antibodies block, CDR H3 mean RMSD in Angstrom; excludes later Nanobodies and TCR blocks. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-041","kind":"claim","name":"Reported Mean CDR H3 RMSD for Chai-1","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Antibody loop structure prediction"]},"source_ids":["ibex-2025"],"links":[{"relation":"subject","target_id":"lit-b3-041"}],"attributes":{"field":"attributes.printed_value","value":"2.65","source_locator":"Table 1, Antibodies / Chai-1 row, CDR H3 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.428536+00:00","notes":"First Antibodies block, CDR H3 mean RMSD in Angstrom; excludes later Nanobodies and TCR blocks. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-042","kind":"claim","name":"Reported Pearson R for DEELIG","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand binding affinity prediction"]},"source_ids":["deelig-2021"],"links":[{"relation":"subject","target_id":"lit-b3-042"}],"attributes":{"field":"attributes.printed_value","value":"0.889","source_locator":"Table 2, DEELIG row, PDBbind v2016 column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:55.586Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b3-043","kind":"claim","name":"Reported Pearson R for TOPBP (Complex)","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand binding affinity prediction"]},"source_ids":["deelig-2021"],"links":[{"relation":"subject","target_id":"lit-b3-043"}],"attributes":{"field":"attributes.printed_value","value":"0.861","source_locator":"Table 2, TOPBP (Complex) row, PDBbind v2016 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.429669+00:00","notes":"TOPBP Complex reference row; PDBbind v2016 core-set Pearson correlation. Third-party comparator with cited reference; do not infer an independent new run from table inclusion. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-044","kind":"claim","name":"Reported RMSD ≤1 Å and PB-valid success for MolAS","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Physically valid protein–ligand pose selection"]},"source_ids":["molas-2026"],"links":[{"relation":"subject","target_id":"lit-b3-044"}],"attributes":{"field":"attributes.printed_value","value":"36.69","source_locator":"Table 3, PoseBusters / Mixed / AutoDock row, MolAS success column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.432565+00:00","notes":"PoseBusters, Mixed, AutoDock row within jointly trained with/without relaxation block. Selected RMSD <=1 Angstrom AND PB-valid group; five-fold average success percentage, not <=2 Angstrom. Inline bold digit nodes joined in original order. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-045","kind":"claim","name":"Reported RMSD ≤1 Å and PB-valid success for Single best solver","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Physically valid protein–ligand pose selection"]},"source_ids":["molas-2026"],"links":[{"relation":"subject","target_id":"lit-b3-045"}],"attributes":{"field":"attributes.printed_value","value":"34.34","source_locator":"Table 3, PoseBusters / Mixed / AutoDock row, SBS success column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.435454+00:00","notes":"PoseBusters, Mixed, AutoDock row within jointly trained with/without relaxation block. Selected RMSD <=1 Angstrom AND PB-valid group; five-fold average success percentage, not <=2 Angstrom. Inline bold digit nodes joined in original order. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-046","kind":"claim","name":"Reported Docked frames best-matched RMSD <3 Å for AutoDock Vina holo","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Intrinsically disordered protein ensemble docking"]},"source_ids":["ensemble-idp-docking-2025"],"links":[{"relation":"subject","target_id":"lit-b3-046"}],"attributes":{"field":"attributes.printed_value","value":"27.96","source_locator":"Table 2, Ligand 47 row, AutoDock Vina Holo Docking column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:56.275Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b3-047","kind":"claim","name":"Reported Docked frames best-matched RMSD <3 Å for DiffDock holo","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Intrinsically disordered protein ensemble docking"]},"source_ids":["ensemble-idp-docking-2025"],"links":[{"relation":"subject","target_id":"lit-b3-047"}],"attributes":{"field":"attributes.printed_value","value":"21.32","source_locator":"Table 2, Ligand 47 row, DiffDock Holo Docking column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:56.275Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b3-048","kind":"claim","name":"Reported Pearson R for AK-score-ensemble","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand binding affinity scoring"]},"source_ids":["akscore-2020"],"links":[{"relation":"subject","target_id":"lit-b3-048"}],"attributes":{"field":"attributes.printed_value","value":"0.812","source_locator":"Table 2, AK-score-ensemble / learning rate 0.0007 row, Scoring Pearson (R) column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.436853+00:00","notes":"CASF-2016 scoring Pearson R with learning rate0.0007. Single-model versus ensemble blocks kept distinct; ranking/docking scores not substituted. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-049","kind":"claim","name":"Reported Pearson R for AK-score-single","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand binding affinity scoring"]},"source_ids":["akscore-2020"],"links":[{"relation":"subject","target_id":"lit-b3-049"}],"attributes":{"field":"attributes.printed_value","value":"0.759","source_locator":"Table 2, AK-score-single / learning rate 0.0007 row, Scoring Pearson (R) column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.437894+00:00","notes":"CASF-2016 scoring Pearson R with learning rate0.0007. Single-model versus ensemble blocks kept distinct; ranking/docking scores not substituted. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-050","kind":"claim","name":"Reported Pearson R for PMF + ECFP + PF (LightGBM)","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand binding energy prediction"]},"source_ids":["fingerprint-scoring-2022"],"links":[{"relation":"subject","target_id":"lit-b3-050"}],"attributes":{"field":"attributes.printed_value","value":"0.79","source_locator":"Table 1, PMF + ECFP + PF / LightGBM row, R column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.439460+00:00","notes":"Resolved two-row model rowspan: last LightGBM row is PMF+ECFP+PF; first LASSO row is PMF. Pearson R, not RMSE. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-051","kind":"claim","name":"Reported Pearson R for PMF (LASSO)","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand binding energy prediction"]},"source_ids":["fingerprint-scoring-2022"],"links":[{"relation":"subject","target_id":"lit-b3-051"}],"attributes":{"field":"attributes.printed_value","value":"0.67","source_locator":"Table 1, PMF / LASSO row, R column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.440812+00:00","notes":"Resolved two-row model rowspan: last LightGBM row is PMF+ECFP+PF; first LASSO row is PMF. Pearson R, not RMSE. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b4-001","kind":"claim","name":"Reported AUROC for ARSENAL+ChromBPNet","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["regulatory-variant scoring"]},"source_ids":["arsenal-regulatory-dna-2026"],"links":[{"relation":"subject","target_id":"lit-b4-001"}],"attributes":{"field":"attributes.printed_value","value":"0.896","source_locator":"Table 1, Yoruban LCL dsQTLs section, ARSENAL+ChromBPNet row, AUROC column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"Yoruban LCL dsQTLs; ARSENAL+ChromBPNet AUROC Transcription verified; experimental claims not independently reproduced."}}} {"id":"claim-lit-b4-002","kind":"claim","name":"Reported AUROC for PlantCAD2","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["cross-species conservation prediction"]},"source_ids":["plantcad2-2025"],"links":[{"relation":"subject","target_id":"lit-b4-002"}],"attributes":{"field":"attributes.printed_value","value":"0.725","source_locator":"Table 1, Cross-species evolutionary conservation > Conservation within Andropogoneae (Genome-wide) row, PlantCAD2 AUROC entry","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"First comparison entry is PlantCAD2; AUROC is 0.725 versus comparator 0.691. Transcription verified; experimental claims not independently reproduced."}}} {"id":"claim-lit-b4-003","kind":"claim","name":"Reported accuracy for Stacking-Auto","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["enhancer prediction"]},"source_ids":["hi-enhancer-2025"],"links":[{"relation":"subject","target_id":"lit-b4-003"}],"attributes":{"field":"attributes.printed_value","value":"80.50","source_locator":"Table 2, Ours (Stacking-Auto) row, Accuracy column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"Ours row, Accuracy column; original source method is the Stacking-Auto stage. Transcription verified; experimental claims not independently reproduced."}}} {"id":"claim-lit-b4-004","kind":"claim","name":"Reported AUROC for position-aware CNN","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["enhancer prediction"]},"source_ids":["enhancer-position-encoding-2024"],"links":[{"relation":"subject","target_id":"lit-b4-004"}],"attributes":{"field":"attributes.printed_value","value":"0.94","source_locator":"Table 2, Human section, CNN row, AUC column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"Human dataset row, CNN, AUC column. Transcription verified; experimental claims not independently reproduced."}}} {"id":"claim-lit-b4-005","kind":"claim","name":"Reported F1 for ADAR-GPT continual","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["A-to-I RNA editing site prediction"]},"source_ids":["adar-gpt-editing-2026"],"links":[{"relation":"subject","target_id":"lit-b4-005"}],"attributes":{"field":"attributes.printed_value","value":"0.763","source_locator":"Table 2, Adar-GPT (continual) row, F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"Adar-GPT continual row; XML inline decimal reordered by parser, original text verified separately. Transcription verified; experimental claims not independently reproduced."}}} {"id":"claim-lit-b4-006","kind":"claim","name":"Reported sequence recovery for R3Design","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA sequence design"]},"source_ids":["r3design-2025"],"links":[{"relation":"subject","target_id":"lit-b4-006"}],"attributes":{"field":"attributes.printed_value","value":"43.27","source_locator":"Table 3, R3Design row, Recovery (%) > Rfam column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"R3Design row, first Recovery column Rfam; 43.27 plus/minus0.56. Transcription verified; experimental claims not independently reproduced."}}} {"id":"claim-lit-b4-007","kind":"claim","name":"Reported AUROC for CUPID Data-aug-Avg","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["non-coding RNA pairwise interaction prediction"]},"source_ids":["cupid-rna-interactions-2026"],"links":[{"relation":"subject","target_id":"lit-b4-007"}],"attributes":{"field":"attributes.printed_value","value":"0.919","source_locator":"Table 1, CUPID > Data-aug-Avg row, AUROC column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"CUPID section Data-aug-Avg row; AUROC not AUPRC. Transcription verified; experimental claims not independently reproduced."}}} {"id":"claim-lit-b4-008","kind":"claim","name":"Reported AUROC for ProteinBERT LLM-encoding model","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["mRNA-protein interaction prediction"]},"source_ids":["mrna-protein-diversity-2026"],"links":[{"relation":"subject","target_id":"lit-b4-008"}],"attributes":{"field":"attributes.printed_value","value":"71.5","source_locator":"Table 2, RBP-aware test set row, auROC (%) column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:57.257Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b4-009","kind":"claim","name":"Reported AUROC for ESM2 650M","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["human-versus-viral protein classification"]},"source_ids":["viral-immune-mimicry-2025"],"links":[{"relation":"subject","target_id":"lit-b4-009"}],"attributes":{"field":"attributes.printed_value","value":"99.67","source_locator":"Table 1, ESM2 650M row, AUC (%) column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:57.274Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b4-010","kind":"claim","name":"Reported AUROC for ProtT5 embeddings + ensemble classifier","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["protein-protein binding-site prediction"]},"source_ids":["protein-binding-sites-2023"],"links":[{"relation":"subject","target_id":"lit-b4-010"}],"attributes":{"field":"attributes.printed_value","value":"0.810","source_locator":"Table 2, Dset_448 section, ProtT5 row, AUROC column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"Dset_448 block; ProtT5 AUROC, downstream ensemble retained in protocol. Transcription verified; experimental claims not independently reproduced."}}} {"id":"claim-lit-b4-011","kind":"claim","name":"Reported AUROC for CLAPE-SMB with ESM-2","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["protein-small molecule binding-site prediction"]},"source_ids":["clape-smb-2024"],"links":[{"relation":"subject","target_id":"lit-b4-011"}],"attributes":{"field":"attributes.printed_value","value":"0.917","source_locator":"Table 5, ESM-2 / SJC row, AUROC column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"ESM-2 on SJC AUROC. Transcription verified; experimental claims not independently reproduced."}}} {"id":"claim-lit-b4-012","kind":"claim","name":"Reported AUPRC for Vaxign-DL + ESM","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["vaccine-antigen candidate prediction"]},"source_ids":["vaxign-esm-2024"],"links":[{"relation":"subject","target_id":"lit-b4-012"}],"attributes":{"field":"attributes.printed_value","value":"0.92","source_locator":"Table 2, 4 Layers row, AUPRC column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"Source spells 4 Layerss; AUPRC0.92±0.013. Transcription verified; experimental claims not independently reproduced."}}} {"id":"claim-lit-b4-013","kind":"claim","name":"Reported AUROC for scGPT + residual geometry","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["gene-regulatory signal prediction"]},"source_ids":["single-cell-residual-geometry-2026"],"links":[{"relation":"subject","target_id":"lit-b4-013"}],"attributes":{"field":"attributes.printed_value","value":"0.677","source_locator":"Table 4, Immune row, scGPT > +geom AUROC column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"Immune row, scGPT +geom (second numeric column), not Geneformer or delta. Transcription verified; experimental claims not independently reproduced."}}} {"id":"claim-lit-b4-014","kind":"claim","name":"Reported F1 for GREmLN","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["cell-type annotation"]},"source_ids":["gremln-2026"],"links":[{"relation":"subject","target_id":"lit-b4-014"}],"attributes":{"field":"attributes.printed_value","value":"0.937","source_locator":"Table 2, Cell type annotation(zero-shot), Non-immune cells, F1 row, GREmLN column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:57.502Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b4-015","kind":"claim","name":"Reported F1 for Cell-DINO ViT-L","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["protein localization classification"]},"source_ids":["cell-dino-2025"],"links":[{"relation":"subject","target_id":"lit-b4-015"}],"attributes":{"field":"attributes.printed_value","value":"65.5","source_locator":"Table 2, HPA-FoV section, Cell-DINO row, PL column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"HPA-FoV Cell-DINO PL column (protein localisation), not CL. Transcription verified; experimental claims not independently reproduced."}}} {"id":"claim-lit-b4-016","kind":"claim","name":"Reported precision at 50% recall for scGen","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["differentially expressed gene identification"]},"source_ids":["insilico-perturbation-auprc-2025"],"links":[{"relation":"subject","target_id":"lit-b4-016"}],"attributes":{"field":"attributes.printed_value","value":"0.91","source_locator":"Table 3, CD14+Mono section, scGen row, Precision at 50% Recall column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"CD14+Mono scGen; precision at50%recall. Transcription verified; experimental claims not independently reproduced."}}} {"id":"claim-lit-b4-017","kind":"claim","name":"Reported F1 for TCINet + HTRS","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["pathogen detection"]},"source_ids":["metagenomic-pathogens-2025"],"links":[{"relation":"subject","target_id":"lit-b4-017"}],"attributes":{"field":"attributes.printed_value","value":"0.84","source_locator":"Table 3, MetaHIT dataset section, TCINet + HTRS (Ours) row, F1-score column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"MetaHIT block TCINet+HTRS F1-score. Transcription verified; experimental claims not independently reproduced."}}} {"id":"claim-lit-b4-018","kind":"claim","name":"Reported accuracy for DETIRE","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["viral sequence detection"]},"source_ids":["detire-viral-metagenomes-2023"],"links":[{"relation":"subject","target_id":"lit-b4-018"}],"attributes":{"field":"attributes.printed_value","value":"0.8772","source_locator":"Table 1, Accuracy row, DETIRE column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:58.392Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b4-019","kind":"claim","name":"Reported accuracy for PC-mer + LR","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["metagenomic genus classification"]},"source_ids":["pc-mer-2024"],"links":[{"relation":"subject","target_id":"lit-b4-019"}],"attributes":{"field":"attributes.printed_value","value":"96.95","source_locator":"Table 3, AMP section, PC-mer + LR k=8 row, Accuracy (%) column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"AMP block PC-mer+LR section, k=8, first numeric value after k is Accuracy. Transcription verified; experimental claims not independently reproduced."}}} {"id":"claim-lit-b4-020","kind":"claim","name":"Reported accuracy for MDL4Microbiome","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["microbiome disease-state classification"]},"source_ids":["mdl4microbiome-2022"],"links":[{"relation":"subject","target_id":"lit-b4-020"}],"attributes":{"field":"attributes.printed_value","value":"0.97","source_locator":"Table 3, CRC row, MDL4Microbiome column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:58.492Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b4-021","kind":"claim","name":"Reported Pearson correlation for binding-affinity meta-model","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["protein-ligand binding affinity prediction"]},"source_ids":["ligand-affinity-meta-model-2024"],"links":[{"relation":"subject","target_id":"lit-b4-021"}],"attributes":{"field":"attributes.printed_value","value":"0.777","source_locator":"Table 4, Meta-models row, CASF-2016 Benchmark > PCC column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"Meta-models CASF-2016 PCC. Confirmed XML training-set rowspan inherits preceding row, so0.777 maps to PCC. Transcription verified; experimental claims not independently reproduced."}}} {"id":"claim-lit-b4-022","kind":"claim","name":"Reported AUROC for DeepInterAware","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["antigen-antibody HIV neutralization prediction"]},"source_ids":["deepinteraware-2025"],"links":[{"relation":"subject","target_id":"lit-b4-022"}],"attributes":{"field":"attributes.printed_value","value":"0.826","source_locator":"Table 2, Ab Unseen section, DeepInterAware row, AUROC column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"Ab Unseen block DeepInterAware AUROC0.826±0.017, not Ag Unseen. Transcription verified; experimental claims not independently reproduced."}}} {"id":"claim-lit-b4-023","kind":"claim","name":"Reported AUROC for TransBind","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["transcription-factor DNA binding-site prediction"]},"source_ids":["transbind-2026"],"links":[{"relation":"subject","target_id":"lit-b4-023"}],"attributes":{"field":"attributes.printed_value","value":"0.9508","source_locator":"Table 2, TransBind row, AUROC column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:58.585Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b4-024","kind":"claim","name":"Reported AUROC for ESM2_AMPS","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["protein-protein interaction prediction"]},"source_ids":["esm2-amp-2025"],"links":[{"relation":"subject","target_id":"lit-b4-024"}],"attributes":{"field":"attributes.printed_value","value":"0.68","source_locator":"Table 4, ESM2_AMPS row, AUROC column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:58.625Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"clape-smb-2024","kind":"source","name":"Protein-small molecule binding site prediction based on a pre-trained protein language model with contrastive learning","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["proteins-complexes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11542454/","version":"version of record","retrieved_at":"2026-09-15T23:37:05Z","doi":"10.1186/s13321-024-00920-2","publication_status":"peer_reviewed","year":2024,"artifact_sha256":"215919244c3dd2dfb0b55fce91c211430fd8d4aee4bb28bd03eab9f4feb73e62","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11542454/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:06Z","legacy_paper":{"id":"clape-smb-2024","title":"Protein-small molecule binding site prediction based on a pre-trained protein language model with contrastive learning","year":2024,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11542454/","primary_domain":"proteins-complexes","retrieved_utc":"2026-09-15T23:37:05Z","notes":"Primary full text verified via Europe PMC fullTextXML; venue: Journal of Cheminformatics; PMC ID: PMC11542454. ESM-2 feature extractor embedded in CLAPE-SMB; score belongs to combined downstream system.","doi":"10.1186/s13321-024-00920-2"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"clathrin-plm-2025","kind":"source","name":"Advancing the accuracy of clathrin protein prediction through multi-source protein language models","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["proteins-complexes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12238356/","version":"journal full text in PMC","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.1038/s41598-025-08510-4","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"2edc86b25707c1b737d26117093ce8d856e79cc5d0b335f27c1c341f887f1c7e","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12238356/fullTextXML","artifact_retrieved_at":"2026-09-16T10:38:57.558194+00:00","legacy_paper":{"id":"clathrin-plm-2025","title":"Advancing the accuracy of clathrin protein prediction through multi-source protein language models","year":2025,"publication_status":"peer_reviewed","version":"journal full text in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12238356/","primary_domain":"proteins-complexes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.1038/s41598-025-08510-4","notes":"Numeric result checked against Table 2 in primary full-text XML; journal/source: Scientific Reports."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"cobra-rna-binding-2026","kind":"source","name":"CoBRA: compound binding site prediction using RNA language model","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["rna-transcriptomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12790621/","version":"journal full text in PMC","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.1093/bib/bbaf713","publication_status":"peer_reviewed","year":2026,"artifact_sha256":"8c6a6f00f5fa5f62acf301a66e9e6fa9ef11c7a05ad9b7447d2ade2ce8eba793","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12790621/fullTextXML","artifact_retrieved_at":"2026-09-16T10:38:57.558197+00:00","legacy_paper":{"id":"cobra-rna-binding-2026","title":"CoBRA: compound binding site prediction using RNA language model","year":2026,"publication_status":"peer_reviewed","version":"journal full text in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12790621/","primary_domain":"rna-transcriptomes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.1093/bib/bbaf713","notes":"Numeric result checked against Table 2 in primary full-text XML; journal/source: Briefings in Bioinformatics."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"codonbert-vaccines-2024","kind":"source","name":"CodonBERT large language model for mRNA vaccines","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["rna-transcriptomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11368176/","version":"journal full text in PMC","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.1101/gr.278870.123","publication_status":"peer_reviewed","year":2024,"artifact_sha256":"2968073753e6d44feff9c08b131edf23145e95b171434539dddf77bb92847033","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11368176/fullTextXML","artifact_retrieved_at":"2026-09-16T10:38:57.558201+00:00","legacy_paper":{"id":"codonbert-vaccines-2024","title":"CodonBERT large language model for mRNA vaccines","year":2024,"publication_status":"peer_reviewed","version":"journal full text in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11368176/","primary_domain":"rna-transcriptomes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.1101/gr.278870.123","notes":"Numeric result checked against Table 2. in primary full-text XML; journal/source: Genome Research."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"cupid-rna-interactions-2026","kind":"source","name":"Computational understanding of non-coding RNA pairwise interactions","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["rna-transcriptomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12957212/","version":"PMC archival version PMC12957212.1","retrieved_at":"2026-09-15T23:37:05Z","doi":"10.3389/frai.2026.1749205","publication_status":"peer_reviewed","year":2026,"artifact_sha256":"0e6719410b390ee9c4858bb9321042851100fb74df3aa109bf2af2b8aaff7ac1","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12957212/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:06Z","legacy_paper":{"id":"cupid-rna-interactions-2026","title":"Computational understanding of non-coding RNA pairwise interactions","year":2026,"publication_status":"peer_reviewed","version":"PMC archival version PMC12957212.1","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12957212/","primary_domain":"rna-transcriptomes","retrieved_utc":"2026-09-15T23:37:05Z","notes":"Primary full text verified via Europe PMC fullTextXML; venue: Frontiers in Artificial Intelligence; PMC ID: PMC12957212. RNA-RNA pairwise interaction predictor; not a foundation model.","doi":"10.3389/frai.2026.1749205"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"cyaprombert-2022","kind":"source","name":"TSSNote-CyaPromBERT: Development of an integrated platform for highly accurate promoter prediction and visualization of Synechococcus sp. and Synechocystis sp. through a state-of-the-art natural language processing model BERT","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC9745317/","version":"version of record","retrieved_at":"2026-09-15T23:25:00Z","doi":"10.3389/fgene.2022.1067562","publication_status":"peer_reviewed","year":2022,"artifact_sha256":"74278ccd77b2bc00a3f4434546545e8bdec8b0652a0e5d1862ec0f91decccd8d","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC9745317/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:16.544033+00:00","legacy_paper":{"id":"cyaprombert-2022","title":"TSSNote-CyaPromBERT: Development of an integrated platform for highly accurate promoter prediction and visualization of Synechococcus sp. and Synechocystis sp. through a state-of-the-art natural language processing model BERT","year":2022,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC9745317/","primary_domain":"microbes-communities","retrieved_utc":"2026-09-15T23:25:00Z","notes":"Primary full text via Europe PMC XML; venue: Frontiers in Genetics; PMC ID: PMC9745317.","doi":"10.3389/fgene.2022.1067562"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"dart-eval-regulatory-2024","kind":"source","name":"DART-Eval: A Comprehensive DNA Language Model Evaluation Benchmark on Regulatory DNA","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["dna-genomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://proceedings.neurips.cc/paper_files/paper/2024/file/71998bfc3217ffe1cca1ee084dfadadd-Paper-Datasets_and_Benchmarks_Track.pdf","version":"NeurIPS 2024 Datasets and Benchmarks Track proceedings","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.52202/079017-1981","publication_status":"peer_reviewed","year":2024,"artifact_sha256":"e5aee5b1f7cc6fd961b1d2a131d02cf243b79e091d5e418fbabee7fde9b39b22","artifact_url":"https://proceedings.neurips.cc/paper_files/paper/2024/file/71998bfc3217ffe1cca1ee084dfadadd-Paper-Datasets_and_Benchmarks_Track.pdf","artifact_retrieved_at":"2026-09-16T10:38:57.558203+00:00","legacy_paper":{"id":"dart-eval-regulatory-2024","title":"DART-Eval: A Comprehensive DNA Language Model Evaluation Benchmark on Regulatory DNA","year":2024,"publication_status":"peer_reviewed","version":"NeurIPS 2024 Datasets and Benchmarks Track proceedings","source_url":"https://proceedings.neurips.cc/paper_files/paper/2024/file/71998bfc3217ffe1cca1ee084dfadadd-Paper-Datasets_and_Benchmarks_Track.pdf","primary_domain":"dna-genomes","retrieved_utc":"2026-09-15T23:33:26Z","notes":"Proceedings Table 3, DNABERT-2 Zero-Shot Accuracy 0.876 checked directly; the PMC/arXiv manuscript carries the same printed row.","doi":"10.52202/079017-1981"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"debfold-2024","kind":"source","name":"DEBFold: Computational Identification of RNA Secondary Structures for Sequences across Structural Families Using Deep Learning","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["rna-transcriptomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11094721/","version":"PMC11094721.1","retrieved_at":"2026-09-15T23:25:00Z","doi":"10.1021/acs.jcim.4c00458","publication_status":"peer_reviewed","year":2024,"artifact_sha256":"e8f960eafb7f00edfdd81d4fb75c6de838e9b872b7e18875fc7a5bff2a2f72b3","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11094721/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:16.509290+00:00","legacy_paper":{"id":"debfold-2024","title":"DEBFold: Computational Identification of RNA Secondary Structures for Sequences across Structural Families Using Deep Learning","year":2024,"publication_status":"peer_reviewed","version":"PMC11094721.1","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11094721/","primary_domain":"rna-transcriptomes","retrieved_utc":"2026-09-15T23:25:00Z","notes":"Primary full text via Europe PMC XML; venue: Journal of Chemical Information and Modeling; PMC ID: PMC11094721.","doi":"10.1021/acs.jcim.4c00458"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"deelig-2021","kind":"source","name":"DEELIG: A Deep Learning Approach to Predict Protein-Ligand Binding Affinity","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC8274096/","version":"PMC archival version PMC8274096.1","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.1177/11779322211030364","publication_status":"peer_reviewed","year":2021,"artifact_sha256":"5a7620c18d0622561004e1e25b5cfaf7399e93df3547eeefdd4cf6d300bb8aba","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC8274096/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:55.586Z","legacy_paper":{"id":"deelig-2021","title":"DEELIG: A Deep Learning Approach to Predict Protein-Ligand Binding Affinity","year":2021,"publication_status":"peer_reviewed","version":"PMC archival version PMC8274096.1","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC8274096/","primary_domain":"molecular-interactions","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: Bioinformatics and Biology Insights; PMC ID: PMC8274096.","doi":"10.1177/11779322211030364"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"deepinteraware-2025","kind":"source","name":"DeepInterAware: Deep Interaction Interface‐Aware Network for Improving Antigen‐Antibody Interaction Prediction from Sequence Data","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11967782/","version":"version of record","retrieved_at":"2026-09-15T23:37:05Z","doi":"10.1002/advs.202412533","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"25d3561934965f754d8712ec02b2052e9a3979e433b88ecebd5db14e930f17a1","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11967782/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:06Z","legacy_paper":{"id":"deepinteraware-2025","title":"DeepInterAware: Deep Interaction Interface‐Aware Network for Improving Antigen‐Antibody Interaction Prediction from Sequence Data","year":2025,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11967782/","primary_domain":"molecular-interactions","retrieved_utc":"2026-09-15T23:37:05Z","notes":"Primary full text verified via Europe PMC fullTextXML; venue: Advanced Science; PMC ID: PMC11967782. Neutralization prediction, not generic binding affinity; uncertainty printed in source table.","doi":"10.1002/advs.202412533"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"detire-viral-metagenomes-2023","kind":"source","name":"DETIRE: a hybrid deep learning model for identifying viral sequences from metagenomes","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC10313334/","version":"PMC archival version PMC10313334.1","retrieved_at":"2026-09-15T23:37:05Z","doi":"10.3389/fmicb.2023.1169791","publication_status":"peer_reviewed","year":2023,"artifact_sha256":"9ff7d32758620f7b0b0628425f62abff103ca2e33269ce3763383584bcebfc3c","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC10313334/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:58.392Z","legacy_paper":{"id":"detire-viral-metagenomes-2023","title":"DETIRE: a hybrid deep learning model for identifying viral sequences from metagenomes","year":2023,"publication_status":"peer_reviewed","version":"PMC archival version PMC10313334.1","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC10313334/","primary_domain":"microbes-communities","retrieved_utc":"2026-09-15T23:37:05Z","notes":"Primary full text verified via Europe PMC fullTextXML; venue: Frontiers in Microbiology; PMC ID: PMC10313334. Task-specific viral classifier, included as a microbial metagenomics benchmark.","doi":"10.3389/fmicb.2023.1169791"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"attributes":{"applicability":"proposed","baseline_type":"null-control","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Matched control cells, normalization and evaluation gene set.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["single-cell"]},"id":"discovery-baseline-cell-perturbation-no-change","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-perturbench"}],"name":"Cell perturbation no-change","source_ids":["src-discovery-altoslabs-perturbench"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"biological-procedure","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Receptor preparation, search box, exhaustiveness and conformers.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["molecular-interactions"]},"id":"discovery-baseline-classical-molecular-docking","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-plinder"},{"relation":"model","target_id":"discovery-model-autodock-vina"}],"name":"Classical molecular docking","source_ids":["src-discovery-ccsb-scripps-autodock-vina"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"null-control","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Matched sequence lengths and dinucleotide-preserving shuffle; fix seeds.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["genomics"]},"id":"discovery-baseline-dinucleotide-shuffled-sequence-control","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-dart-eval"}],"name":"Dinucleotide-shuffled sequence control","source_ids":["src-discovery-kundajelab-dart-eval"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"biological-procedure","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Pinned reference genomes, taxonomy and confidence setting.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["microbiome"]},"id":"discovery-baseline-exact-sequence-taxonomic-classification","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-cami-taxonomic-binning"},{"relation":"model","target_id":"discovery-model-kraken-2"}],"name":"Exact-sequence taxonomic classification","source_ids":["src-discovery-derrickwood-kraken2"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"experimental-reference","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Independent biological replicates under matching conditions; not a universal ceiling.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["single-cell"]},"id":"discovery-baseline-experimental-replicate-agreement","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-scperteval"}],"name":"Experimental replicate agreement","source_ids":["src-discovery-virtual-cell-research-community-scperteval"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"mechanistic","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Stoichiometric reconstruction, growth medium, bounds and objective.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["mechanistic-biology"]},"id":"discovery-baseline-flux-balance-prediction","kind":"baseline","links":[{"relation":"model","target_id":"discovery-model-cobrapy"}],"name":"Flux-balance prediction","source_ids":["src-discovery-opencobra-cobrapy"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"simple-statistical","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Training-only motif features and leakage-aware glycan split.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["glycomics"]},"id":"discovery-baseline-glycan-motif-feature-classifier","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-glycanml"}],"name":"Glycan motif feature classifier","source_ids":["src-discovery-bojarlab-glycowork"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"simple-statistical","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Training-only feature fitting; choose k and penalty within training folds.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["genomics"]},"id":"discovery-baseline-k-mer-ridge-regression","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-genomic-benchmarks"}],"name":"k-mer ridge regression","source_ids":["src-discovery-ml-bioinfo-ceitec-genomic-benchmarks"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"biological-procedure","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Adduct, ion mode, library version, mass tolerance and annotation resolution.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["lipidomics"]},"id":"discovery-baseline-lipid-fragmentation-library-match","kind":"baseline","links":[{"relation":"model","target_id":"discovery-model-lipidblast"}],"name":"Lipid fragmentation library match","source_ids":["src-discovery-lipidblast"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"biological-procedure","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Pinned marker database and taxonomic rank.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["microbiome"]},"id":"discovery-baseline-marker-based-microbial-profiling","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-cami-taxonomic-profiling"},{"relation":"model","target_id":"discovery-model-metaphlan"}],"name":"Marker-based microbial profiling","source_ids":["src-discovery-biobakery-metaphlan"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"biological-procedure","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Peak filtering, precursor tolerance, library and candidate set.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["metabolomics"]},"id":"discovery-baseline-mass-spectral-cosine-matching","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-massspecgym-molecule-retrieval"},{"relation":"model","target_id":"discovery-model-matchms"}],"name":"Mass spectral cosine matching","source_ids":["src-discovery-matchms-matchms"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"biological-procedure","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Pinned motif library, background frequencies and strand convention.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["genomics"]},"id":"discovery-baseline-motif-scanning","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-dart-eval"},{"relation":"model","target_id":"discovery-model-fimo"}],"name":"Motif scanning","source_ids":["src-discovery-meme"],"status":"discovered"} {"attributes":{"applicability":"source_supported","baseline_type":"simple-statistical","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Task-specific training split and predictor head.","scope_note":"One Hot is an explicitly reported comparator in official TAPE task tables. This record does not imply the same baseline protocol suits every protein task."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["protein-function"]},"id":"discovery-baseline-one-hot-protein-encoding","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-tape"}],"name":"One-hot protein encoding","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"biological-procedure","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Pinned sequence database, MSA construction and score threshold.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["protein-function"]},"id":"discovery-baseline-profile-hmm-sequence-search","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-tape-remote-homology-detection"},{"relation":"model","target_id":"discovery-model-hh-suite"}],"name":"Profile-HMM sequence search","source_ids":["src-discovery-soedinglab-hh-suite"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"biological-procedure","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Training reference set and homology leakage controls.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["protein-function"]},"id":"discovery-baseline-protein-homology-transfer","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-cafa"},{"relation":"model","target_id":"discovery-model-mmseqs2"}],"name":"Protein homology transfer","source_ids":["src-discovery-soedinglab-mmseqs2"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"learned-specialist","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Pinned backbone, checkpoint and sampling temperature.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["protein-structure"]},"id":"discovery-baseline-protein-sequence-recovery-specialist","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-proteinbench"},{"relation":"model","target_id":"discovery-model-proteinmpnn"}],"name":"Protein sequence recovery specialist","source_ids":["src-discovery-dauparas-proteinmpnn"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"null-control","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Matched candidate edge universe, edge density and seed.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["biological-networks"]},"id":"discovery-baseline-random-regulatory-network","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-beeline"}],"name":"Random regulatory network","source_ids":["src-discovery-murali-group-beeline"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"biological-procedure","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Temperature, thermodynamic parameter set and pseudoknot policy.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["rna"]},"id":"discovery-baseline-rna-minimum-free-energy-folding","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-beacon"},{"relation":"model","target_id":"discovery-model-viennarna-rnafold"}],"name":"RNA minimum-free-energy folding","source_ids":["src-discovery-viennarna-viennarna"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"simple-statistical","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Training-only GC/codon composition features and held-out split.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["rna"]},"id":"discovery-baseline-rna-sequence-composition-regression","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-mrnabench"}],"name":"RNA sequence-composition regression","source_ids":["src-discovery-morrislab-mrnabench"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"simple-statistical","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Frozen features and donor-disjoint folds; training-only regularization.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["spatial-omics"]},"id":"discovery-baseline-spatial-expression-ridge-regression","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-hest-benchmark"}],"name":"Spatial expression ridge regression","source_ids":["src-discovery-mahmoodlab-hest"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"learned-specialist","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Genome assembly, transcript context and model release.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["genomics"]},"id":"discovery-baseline-specialist-splicing-predictor","kind":"baseline","links":[{"relation":"model","target_id":"discovery-model-spliceai"}],"name":"Specialist splicing predictor","source_ids":["src-discovery-illumina-spliceai"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"simple-statistical","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Training-only expression mean with explicit perturbation averaging.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["single-cell"]},"id":"discovery-baseline-training-perturbation-mean","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-perturbench"}],"name":"Training perturbation mean","source_ids":["src-discovery-altoslabs-perturbench"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"learned-specialist","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Expression normalization, regulator list and training cells.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["biological-networks"]},"id":"discovery-baseline-tree-ensemble-regulatory-inference","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-beeline"},{"relation":"model","target_id":"discovery-model-genie3"}],"name":"Tree-ensemble regulatory inference","source_ids":["src-discovery-aertslab-genie3"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"null-control","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Same cells, preprocessing and metrics as integrated embeddings.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["single-cell"]},"id":"discovery-baseline-unintegrated-expression-reference","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-scib"}],"name":"Unintegrated expression reference","source_ids":["src-discovery-theislab-scib"],"status":"discovered"} {"attributes":{"entity_level":"evaluator","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Genome reconstruction and taxonomic assignment evaluation","version":null,"profile":{"summary":"AMBER scores metagenomic binning and taxonomic assignments against a supplied gold standard.","sections":[{"title":"Procedure","body":"Provide predicted and reference sequence assignments in the documented binning format. AMBER reports bin-level and sample-level summaries and comparative plots; the reference and sample identifiers must match.","source_ids":["src-discovery-cami-challenge-amber"],"source_locator":"README: introduction; User guide / Input; Metrics computed per bin and per sample"}],"facts":[{"label":"Record type","value":"Evaluator","source_ids":["src-discovery-cami-challenge-amber"],"source_locator":"README: introduction; User guide / Input; Metrics computed per bin and per sample"},{"label":"Inputs","value":"Predicted bin assignments and gold-standard assignments","source_ids":["src-discovery-cami-challenge-amber"],"source_locator":"README: introduction; User guide / Input; Metrics computed per bin and per sample"},{"label":"Outputs and assessment","value":"Binning accuracy and completeness summaries","source_ids":["src-discovery-cami-challenge-amber"],"source_locator":"README: introduction; User guide / Input; Metrics computed per bin and per sample"}],"strengths":[{"text":"Separates individual-bin quality from sample-wide performance.","source_ids":["src-discovery-cami-challenge-amber"],"source_locator":"README: introduction; User guide / Input; Metrics computed per bin and per sample"}],"limitations":[{"text":"An evaluator does not define the held-out community, database version or tool fitting procedure.","source_ids":["src-discovery-cami-challenge-amber"],"source_locator":"README: introduction; User guide / Input; Metrics computed per bin and per sample"}],"diagram":{"title":"Procedure overview","steps":["Sequence assignments","Match reference bins","Compute bin and sample metrics"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-cami-challenge-amber"],"source_locator":"README: introduction; User guide / Input; Metrics computed per bin and per sample"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Genome reconstruction and taxonomic assignment evaluation","facets":{"areas":["microbiome"]},"id":"discovery-benchmark-amber","kind":"benchmark","links":[],"name":"AMBER","source_ids":["src-discovery-cami-challenge-amber"],"status":"discovered"} {"attributes":{"entity_level":"suite","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Three-dimensional molecular learning tasks","version":null,"profile":{"summary":"ATOM3D provides datasets and tooling for learning from three-dimensional molecular structures.","sections":[{"title":"Procedure","body":"Select a molecular task and its dataset, load coordinates and labels, and use the task-specific splitting and evaluation instructions. The package supports structure files and LMDB datasets.","source_ids":["src-discovery-drorlab-atom3d"],"source_locator":"README: Features; Usage / Downloading a dataset; Reference"}],"facts":[{"label":"Record type","value":"Task suite and data tooling","source_ids":["src-discovery-drorlab-atom3d"],"source_locator":"README: Features; Usage / Downloading a dataset; Reference"},{"label":"Inputs","value":"Molecular structures and task-specific labels","source_ids":["src-discovery-drorlab-atom3d"],"source_locator":"README: Features; Usage / Downloading a dataset; Reference"},{"label":"Outputs and assessment","value":"Task-specific molecular predictions","source_ids":["src-discovery-drorlab-atom3d"],"source_locator":"README: Features; Usage / Downloading a dataset; Reference"}],"strengths":[{"text":"Reusable loading, filtering and splitting utilities support consistent structural-data handling.","source_ids":["src-discovery-drorlab-atom3d"],"source_locator":"README: Features; Usage / Downloading a dataset; Reference"}],"limitations":[{"text":"The suite name alone does not specify a dataset, split, metric or permitted structural information.","source_ids":["src-discovery-drorlab-atom3d"],"source_locator":"README: Features; Usage / Downloading a dataset; Reference"}],"diagram":{"title":"Procedure overview","steps":["Select task dataset","Load molecular structures","Fit task predictor","Score task outputs"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-drorlab-atom3d"],"source_locator":"README: Features; Usage / Downloading a dataset; Reference"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Three-dimensional molecular learning tasks","facets":{"areas":["molecular-interactions"]},"id":"discovery-benchmark-atom3d","kind":"benchmark","links":[],"name":"ATOM3D","source_ids":["src-discovery-drorlab-atom3d"],"status":"discovered"} {"attributes":{"entity_level":"suite","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"RNA structure, function and engineering tasks","version":null,"profile":{"summary":"BEACON evaluates RNA representations across structure, function and engineering tasks.","sections":[{"title":"Procedure","body":"Choose the named task directory and matching fine-tuning script. Inputs and prediction heads differ between sequence classification, base-pair maps, degradation, translation and CRISPR-related tasks.","source_ids":["src-discovery-terry-r123-rnabenchmark"],"source_locator":"README: Tasks and Datasets; Usage / Finetuning"}],"facts":[{"label":"Record type","value":"Task suite","source_ids":["src-discovery-terry-r123-rnabenchmark"],"source_locator":"README: Tasks and Datasets; Usage / Finetuning"},{"label":"Inputs","value":"RNA sequences with task-specific labels","source_ids":["src-discovery-terry-r123-rnabenchmark"],"source_locator":"README: Tasks and Datasets; Usage / Finetuning"},{"label":"Outputs and assessment","value":"Classification, regression or structural predictions","source_ids":["src-discovery-terry-r123-rnabenchmark"],"source_locator":"README: Tasks and Datasets; Usage / Finetuning"}],"strengths":[{"text":"Makes multiple RNA task types available through a shared model evaluation codebase.","source_ids":["src-discovery-terry-r123-rnabenchmark"],"source_locator":"README: Tasks and Datasets; Usage / Finetuning"}],"limitations":[{"text":"Task scripts and data revisions must be pinned separately; one RNA task score cannot represent all RNA capabilities.","source_ids":["src-discovery-terry-r123-rnabenchmark"],"source_locator":"README: Tasks and Datasets; Usage / Finetuning"}],"diagram":{"title":"Procedure overview","steps":["Choose RNA task","Load released split","Fine-tune task head","Evaluate task output"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-terry-r123-rnabenchmark"],"source_locator":"README: Tasks and Datasets; Usage / Finetuning"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"RNA structure, function and engineering tasks","facets":{"areas":["rna"]},"id":"discovery-benchmark-beacon","kind":"benchmark","links":[],"name":"BEACON","source_ids":["src-discovery-terry-r123-rnabenchmark"],"status":"discovered"} {"attributes":{"entity_level":"suite","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Gene regulatory network inference","version":null,"profile":{"summary":"BEELINE evaluates gene regulatory network inference from single-cell expression data.","sections":[{"title":"Procedure","body":"Run a selected inference algorithm, export its ranked regulatory edges and compare those edges with a supplied reference network. The framework separates execution, evaluation and plotting.","source_ids":["src-discovery-murali-group-beeline"],"source_locator":"README: Usage / BLRunner.py, BLEvaluator.py, BLPlotter.py; Configuration"}],"facts":[{"label":"Record type","value":"Benchmark framework and evaluator","source_ids":["src-discovery-murali-group-beeline"],"source_locator":"README: Usage / BLRunner.py, BLEvaluator.py, BLPlotter.py; Configuration"},{"label":"Inputs","value":"Single-cell expression and a reference regulatory network","source_ids":["src-discovery-murali-group-beeline"],"source_locator":"README: Usage / BLRunner.py, BLEvaluator.py, BLPlotter.py; Configuration"},{"label":"Outputs and assessment","value":"Ranked-edge AUPRC, AUROC and early precision","source_ids":["src-discovery-murali-group-beeline"],"source_locator":"README: Usage / BLRunner.py, BLEvaluator.py, BLPlotter.py; Configuration"}],"strengths":[{"text":"Containerized methods and a common evaluator make methodological comparisons inspectable.","source_ids":["src-discovery-murali-group-beeline"],"source_locator":"README: Usage / BLRunner.py, BLEvaluator.py, BLPlotter.py; Configuration"}],"limitations":[{"text":"The chosen reference network defines what counts as a correct edge; this is not proof that every inferred interaction is causal.","source_ids":["src-discovery-murali-group-beeline"],"source_locator":"README: Usage / BLRunner.py, BLEvaluator.py, BLPlotter.py; Configuration"}],"diagram":{"title":"Procedure overview","steps":["Expression data","Infer ranked edges","Compare reference network","Report edge metrics"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-murali-group-beeline"],"source_locator":"README: Usage / BLRunner.py, BLEvaluator.py, BLPlotter.py; Configuration"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Gene regulatory network inference","facets":{"areas":["biological-networks"]},"id":"discovery-benchmark-beeline","kind":"benchmark","links":[],"name":"BEELINE","source_ids":["src-discovery-murali-group-beeline"],"status":"discovered"} {"attributes":{"entity_level":"suite","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"DNA representations on biological downstream tasks","version":null,"profile":{"summary":"BEND evaluates DNA representations on biologically defined downstream tasks.","sections":[{"title":"Procedure","body":"Generate embeddings for the released genomic intervals, then train the supplied supervised predictor or use the relevant unsupervised scoring procedure. Embeddings are expanded to nucleotide resolution according to each tokenizer.","source_ids":["src-discovery-frederikkemarin-bend"],"source_locator":"README: Tutorial / Data format, Computing embeddings, Evaluating models; FAQ / How are embeddings upsampled?"}],"facts":[{"label":"Record type","value":"Task suite","source_ids":["src-discovery-frederikkemarin-bend"],"source_locator":"README: Tutorial / Data format, Computing embeddings, Evaluating models; FAQ / How are embeddings upsampled?"},{"label":"Inputs","value":"Genomic intervals, genome sequence and task annotations","source_ids":["src-discovery-frederikkemarin-bend"],"source_locator":"README: Tutorial / Data format, Computing embeddings, Evaluating models; FAQ / How are embeddings upsampled?"},{"label":"Outputs and assessment","value":"Gene, regulatory or variant-related task predictions","source_ids":["src-discovery-frederikkemarin-bend"],"source_locator":"README: Tutorial / Data format, Computing embeddings, Evaluating models; FAQ / How are embeddings upsampled?"}],"strengths":[{"text":"Includes one-hot and supervised baselines alongside pretrained representations.","source_ids":["src-discovery-frederikkemarin-bend"],"source_locator":"README: Tutorial / Data format, Computing embeddings, Evaluating models; FAQ / How are embeddings upsampled?"}],"limitations":[{"text":"Tokenizer upsampling and genomic coordinate handling affect what the predictor receives; task-specific splits and metrics remain necessary.","source_ids":["src-discovery-frederikkemarin-bend"],"source_locator":"README: Tutorial / Data format, Computing embeddings, Evaluating models; FAQ / How are embeddings upsampled?"}],"diagram":{"title":"Procedure overview","steps":["Genomic intervals","Compute DNA embeddings","Apply task predictor","Evaluate held-out labels"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-frederikkemarin-bend"],"source_locator":"README: Tutorial / Data format, Computing embeddings, Evaluating models; FAQ / How are embeddings upsampled?"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"DNA representations on biological downstream tasks","facets":{"areas":["genomics"]},"id":"discovery-benchmark-bend","kind":"benchmark","links":[],"name":"BEND","source_ids":["src-discovery-frederikkemarin-bend"],"status":"discovered"} {"attributes":{"entity_level":"challenge","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Protein function prediction challenge","version":null,"profile":{"summary":"CAFA is a time-based challenge for predicting protein function.","sections":[{"title":"Procedure","body":"Submit ontology-term predictions before the deadline. Proteins receiving experimental annotations after submission become evaluation targets for that challenge round.","source_ids":["src-discovery-cafa"],"source_locator":"Official CAFA page: The CAFA Challenge, prediction and annotation-growth timeline"}],"facts":[{"label":"Record type","value":"Prospective challenge","source_ids":["src-discovery-cafa"],"source_locator":"Official CAFA page: The CAFA Challenge, prediction and annotation-growth timeline"},{"label":"Inputs","value":"Protein sequences and ontology-term predictions","source_ids":["src-discovery-cafa"],"source_locator":"Official CAFA page: The CAFA Challenge, prediction and annotation-growth timeline"},{"label":"Outputs and assessment","value":"Agreement with subsequently acquired functional annotations","source_ids":["src-discovery-cafa"],"source_locator":"Official CAFA page: The CAFA Challenge, prediction and annotation-growth timeline"}],"strengths":[{"text":"Uses later experimental annotations to assess predictions made in advance.","source_ids":["src-discovery-cafa"],"source_locator":"Official CAFA page: The CAFA Challenge, prediction and annotation-growth timeline"}],"limitations":[{"text":"Only proteins that acquire suitable annotations enter the assessed set; ontology version, round and scoring rules must be recorded.","source_ids":["src-discovery-cafa"],"source_locator":"Official CAFA page: The CAFA Challenge, prediction and annotation-growth timeline"}],"diagram":{"title":"Procedure overview","steps":["Release target sequences","Submit functions","Accumulate new annotations","Assess predictions"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-cafa"],"source_locator":"Official CAFA page: The CAFA Challenge, prediction and annotation-growth timeline"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Protein function prediction challenge","facets":{"areas":["protein-function"]},"id":"discovery-benchmark-cafa","kind":"benchmark","links":[],"name":"CAFA","source_ids":["src-discovery-cafa"],"status":"discovered"} {"attributes":{"entity_level":"challenge","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Metagenomic assembly, binning and profiling assessment","version":null,"profile":{"summary":"CAMI organizes community assessments of metagenomic assembly, binning and taxonomic profiling.","sections":[{"title":"Procedure","body":"Choose a challenge dataset and task, produce its required output and compare against the corresponding reference. Assembly, genome binning, taxonomic binning and abundance profiling are distinct evaluations.","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"}],"facts":[{"label":"Record type","value":"Challenge family","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"},{"label":"Inputs","value":"Challenge metagenomic data and task-specific submissions","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"},{"label":"Outputs and assessment","value":"Separate assembly, binning and profiling assessments","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"}],"strengths":[{"text":"A shared challenge resource supports comparison across the metagenomic analysis workflow.","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"}],"limitations":[{"text":"Challenge edition, taxonomy and reference corrections matter; records from different editions are not interchangeable.","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"}],"diagram":{"title":"Procedure overview","steps":["Select challenge dataset","Run task method","Submit task output","Compare reference"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Metagenomic assembly, binning and profiling assessment","facets":{"areas":["microbiome"]},"id":"discovery-benchmark-cami","kind":"benchmark","links":[],"name":"CAMI","source_ids":["src-discovery-cami"],"status":"discovered"} {"attributes":{"entity_level":"task","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"genome binning","version":null,"profile":{"summary":"Group metagenomic contigs into candidate genomes.","sections":[{"title":"Procedure","body":"This is the CAMI genome binning component, not the parent suite as a whole. Choose a challenge dataset and task, produce its required output and compare against the corresponding reference. Assembly, genome binning, taxonomic binning and abundance profiling are distinct evaluations.","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"}],"facts":[{"label":"Record type","value":"Component task","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"},{"label":"Inputs","value":"Contigs and predicted genome bins","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"},{"label":"Outputs and assessment","value":"Genome-bin quality against the reference","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"}],"strengths":[{"text":"A shared challenge resource supports comparison across the metagenomic analysis workflow.","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"}],"limitations":[{"text":"Challenge edition, taxonomy and reference corrections matter; records from different editions are not interchangeable.","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"}],"diagram":{"title":"Component evaluation overview","steps":["Contigs and predicted genome bins","CAMI genome binning","Genome-bin quality against the reference"],"caption":"Conceptual component-task overview. Exact data versions, split files and scorer settings must be taken from the source.","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"},"coverage":"limited","gaps":["Component-specific split manifest and complete scoring configuration have not yet been extracted."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Existing catalogue evidence and cited extraction inspected. The specific protocol gaps listed here remain unresolved; numerical source checks do not constitute complete methods review."}}},"description":"genome binning","facets":{"areas":["microbiome"]},"id":"discovery-benchmark-cami-genome-binning","kind":"benchmark","links":[{"relation":"parent","target_id":"discovery-benchmark-cami"},{"relation":"part_of","target_id":"discovery-benchmark-cami"}],"name":"CAMI genome binning","source_ids":["src-discovery-cami"],"status":"discovered"} {"attributes":{"entity_level":"task","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"metagenome assembly","version":null,"profile":{"summary":"Reconstruct sequences from mixed-community reads.","sections":[{"title":"Procedure","body":"This is the CAMI metagenome assembly component, not the parent suite as a whole. Choose a challenge dataset and task, produce its required output and compare against the corresponding reference. Assembly, genome binning, taxonomic binning and abundance profiling are distinct evaluations.","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"}],"facts":[{"label":"Record type","value":"Component task","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"},{"label":"Inputs","value":"Metagenomic reads and assembled contigs","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"},{"label":"Outputs and assessment","value":"Assembly comparison with the challenge reference","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"}],"strengths":[{"text":"A shared challenge resource supports comparison across the metagenomic analysis workflow.","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"}],"limitations":[{"text":"Challenge edition, taxonomy and reference corrections matter; records from different editions are not interchangeable.","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"}],"diagram":{"title":"Component evaluation overview","steps":["Metagenomic reads and assembled contigs","CAMI metagenome assembly","Assembly comparison with the challenge reference"],"caption":"Conceptual component-task overview. Exact data versions, split files and scorer settings must be taken from the source.","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"},"coverage":"limited","gaps":["Component-specific split manifest and complete scoring configuration have not yet been extracted."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Existing catalogue evidence and cited extraction inspected. The specific protocol gaps listed here remain unresolved; numerical source checks do not constitute complete methods review."}}},"description":"metagenome assembly","facets":{"areas":["microbiome"]},"id":"discovery-benchmark-cami-metagenome-assembly","kind":"benchmark","links":[{"relation":"parent","target_id":"discovery-benchmark-cami"},{"relation":"part_of","target_id":"discovery-benchmark-cami"}],"name":"CAMI metagenome assembly","source_ids":["src-discovery-cami"],"status":"discovered"} {"attributes":{"entity_level":"task","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"taxonomic binning","version":null,"profile":{"summary":"Assign individual sequences to taxonomic groups.","sections":[{"title":"Procedure","body":"This is the CAMI taxonomic binning component, not the parent suite as a whole. Choose a challenge dataset and task, produce its required output and compare against the corresponding reference. Assembly, genome binning, taxonomic binning and abundance profiling are distinct evaluations.","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"}],"facts":[{"label":"Record type","value":"Component task","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"},{"label":"Inputs","value":"Reads or contigs with taxonomic assignments","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"},{"label":"Outputs and assessment","value":"Assignment quality by taxonomic rank","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"}],"strengths":[{"text":"A shared challenge resource supports comparison across the metagenomic analysis workflow.","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"}],"limitations":[{"text":"Challenge edition, taxonomy and reference corrections matter; records from different editions are not interchangeable.","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"}],"diagram":{"title":"Component evaluation overview","steps":["Reads or contigs with taxonomic assignments","CAMI taxonomic binning","Assignment quality by taxonomic rank"],"caption":"Conceptual component-task overview. Exact data versions, split files and scorer settings must be taken from the source.","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"},"coverage":"limited","gaps":["Component-specific split manifest and complete scoring configuration have not yet been extracted."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Existing catalogue evidence and cited extraction inspected. The specific protocol gaps listed here remain unresolved; numerical source checks do not constitute complete methods review."}}},"description":"taxonomic binning","facets":{"areas":["microbiome"]},"id":"discovery-benchmark-cami-taxonomic-binning","kind":"benchmark","links":[{"relation":"parent","target_id":"discovery-benchmark-cami"},{"relation":"part_of","target_id":"discovery-benchmark-cami"}],"name":"CAMI taxonomic binning","source_ids":["src-discovery-cami"],"status":"discovered"} {"attributes":{"entity_level":"task","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"taxonomic profiling","version":null,"profile":{"summary":"Estimate which taxa occur and their relative abundance.","sections":[{"title":"Procedure","body":"This is the CAMI taxonomic profiling component, not the parent suite as a whole. Choose a challenge dataset and task, produce its required output and compare against the corresponding reference. Assembly, genome binning, taxonomic binning and abundance profiling are distinct evaluations.","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"}],"facts":[{"label":"Record type","value":"Component task","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"},{"label":"Inputs","value":"Sample-level taxon abundance profiles","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"},{"label":"Outputs and assessment","value":"Presence and abundance agreement by rank","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"}],"strengths":[{"text":"A shared challenge resource supports comparison across the metagenomic analysis workflow.","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"}],"limitations":[{"text":"Challenge edition, taxonomy and reference corrections matter; records from different editions are not interchangeable.","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"}],"diagram":{"title":"Component evaluation overview","steps":["Sample-level taxon abundance profiles","CAMI taxonomic profiling","Presence and abundance agreement by rank"],"caption":"Conceptual component-task overview. Exact data versions, split files and scorer settings must be taken from the source.","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"},"coverage":"limited","gaps":["Component-specific split manifest and complete scoring configuration have not yet been extracted."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Existing catalogue evidence and cited extraction inspected. The specific protocol gaps listed here remain unresolved; numerical source checks do not constitute complete methods review."}}},"description":"taxonomic profiling","facets":{"areas":["microbiome"]},"id":"discovery-benchmark-cami-taxonomic-profiling","kind":"benchmark","links":[{"relation":"parent","target_id":"discovery-benchmark-cami"},{"relation":"part_of","target_id":"discovery-benchmark-cami"}],"name":"CAMI taxonomic profiling","source_ids":["src-discovery-cami"],"status":"discovered"} {"attributes":{"entity_level":"challenge","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Protein interaction docking assessment","version":null,"profile":{"summary":"CAPRI assesses blind predictions of protein-complex structures.","sections":[{"title":"Procedure","body":"Predict the structure of a target complex before its experimental structure is publicly released, then assess submitted models against that structure.","source_ids":["src-discovery-capri"],"source_locator":"Official CAPRI page: Welcome to the CAPRI web site, first paragraph"}],"facts":[{"label":"Record type","value":"Blind structural challenge","source_ids":["src-discovery-capri"],"source_locator":"Official CAPRI page: Welcome to the CAPRI web site, first paragraph"},{"label":"Inputs","value":"Released target information for interacting proteins","source_ids":["src-discovery-capri"],"source_locator":"Official CAPRI page: Welcome to the CAPRI web site, first paragraph"},{"label":"Outputs and assessment","value":"Predicted complex structures evaluated against experiment","source_ids":["src-discovery-capri"],"source_locator":"Official CAPRI page: Welcome to the CAPRI web site, first paragraph"}],"strengths":[{"text":"Unpublished targets support prospective assessment.","source_ids":["src-discovery-capri"],"source_locator":"Official CAPRI page: Welcome to the CAPRI web site, first paragraph"}],"limitations":[{"text":"This family record does not pin a target round, allowed input information or scoring implementation.","source_ids":["src-discovery-capri"],"source_locator":"Official CAPRI page: Welcome to the CAPRI web site, first paragraph"}],"diagram":{"title":"Procedure overview","steps":["Receive complex target","Predict interactions","Reveal reference structure","Assess submitted complexes"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-capri"],"source_locator":"Official CAPRI page: Welcome to the CAPRI web site, first paragraph"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Protein interaction docking assessment","facets":{"areas":["protein-structure"]},"id":"discovery-benchmark-capri","kind":"benchmark","links":[],"name":"CAPRI","source_ids":["src-discovery-capri"],"status":"discovered"} {"attributes":{"entity_level":"challenge","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Community protein structure assessment","version":null,"profile":{"summary":"CASP assesses protein structure predictions through blind community experiments.","sections":[{"title":"Procedure","body":"Predict the released targets for a particular CASP round and category. Assessors compare submissions with experimental structures using the category's published measures.","source_ids":["src-discovery-casp"],"source_locator":"Prediction Center home page: Welcome; blind prediction; data archive and numerical evaluation sections"}],"facts":[{"label":"Record type","value":"Blind structural challenge","source_ids":["src-discovery-casp"],"source_locator":"Prediction Center home page: Welcome; blind prediction; data archive and numerical evaluation sections"},{"label":"Inputs","value":"Target sequences and category-specific permitted information","source_ids":["src-discovery-casp"],"source_locator":"Prediction Center home page: Welcome; blind prediction; data archive and numerical evaluation sections"},{"label":"Outputs and assessment","value":"Structural agreement with withheld experimental targets","source_ids":["src-discovery-casp"],"source_locator":"Prediction Center home page: Welcome; blind prediction; data archive and numerical evaluation sections"}],"strengths":[{"text":"Archived targets, submissions and assessment reports support scrutiny of progress.","source_ids":["src-discovery-casp"],"source_locator":"Prediction Center home page: Welcome; blind prediction; data archive and numerical evaluation sections"}],"limitations":[{"text":"Monomer, assembly, refinement and data-assisted tracks involve different inputs and measures; select the exact round and category.","source_ids":["src-discovery-casp"],"source_locator":"Prediction Center home page: Welcome; blind prediction; data archive and numerical evaluation sections"}],"diagram":{"title":"Procedure overview","steps":["Select round and track","Submit structural predictions","Release experimental targets","Assess structures"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-casp"],"source_locator":"Prediction Center home page: Welcome; blind prediction; data archive and numerical evaluation sections"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Community protein structure assessment","facets":{"areas":["protein-structure"]},"id":"discovery-benchmark-casp","kind":"benchmark","links":[],"name":"CASP","source_ids":["src-discovery-casp"],"status":"discovered"} {"attributes":{"entity_level":"suite","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Human regulatory DNA representation evaluation","version":null,"profile":{"summary":"DART-Eval examines whether DNA representations capture human gene regulation.","sections":[{"title":"Procedure","body":"Evaluate regulatory-element discrimination, motif footprinting, cell-type specificity, quantitative activity and variant effects using the supplied task resources. Distinguish zero-shot scoring, learned probes and fine-tuned models.","source_ids":["src-discovery-kundajelab-dart-eval"],"source_locator":"README: opening; Tasks 1–5 and their zero-shot, probing, fine-tuning and ab initio subsections"}],"facts":[{"label":"Record type","value":"Task suite","source_ids":["src-discovery-kundajelab-dart-eval"],"source_locator":"README: opening; Tasks 1–5 and their zero-shot, probing, fine-tuning and ab initio subsections"},{"label":"Inputs","value":"Human regulatory DNA and task-specific experimental annotations","source_ids":["src-discovery-kundajelab-dart-eval"],"source_locator":"README: opening; Tasks 1–5 and their zero-shot, probing, fine-tuning and ab initio subsections"},{"label":"Outputs and assessment","value":"Regulatory task predictions in explicitly different adaptation regimes","source_ids":["src-discovery-kundajelab-dart-eval"],"source_locator":"README: opening; Tasks 1–5 and their zero-shot, probing, fine-tuning and ab initio subsections"}],"strengths":[{"text":"Includes increasing task difficulty and ab initio comparators.","source_ids":["src-discovery-kundajelab-dart-eval"],"source_locator":"README: opening; Tasks 1–5 and their zero-shot, probing, fine-tuning and ab initio subsections"}],"limitations":[{"text":"Results depend on the adaptation regime and biological task; zero-shot and fine-tuned scores cannot be pooled as the same evaluation.","source_ids":["src-discovery-kundajelab-dart-eval"],"source_locator":"README: opening; Tasks 1–5 and their zero-shot, probing, fine-tuning and ab initio subsections"}],"diagram":{"title":"Procedure overview","steps":["Select regulatory task","Choose adaptation regime","Generate predictions","Score task evidence"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-kundajelab-dart-eval"],"source_locator":"README: opening; Tasks 1–5 and their zero-shot, probing, fine-tuning and ab initio subsections"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Human regulatory DNA representation evaluation","facets":{"areas":["genomics"]},"id":"discovery-benchmark-dart-eval","kind":"benchmark","links":[],"name":"DART-Eval","source_ids":["src-discovery-kundajelab-dart-eval"],"status":"discovered"} {"attributes":{"entity_level":"suite","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Generalisation in protein fitness landscapes","version":null,"profile":{"summary":"FLIP tests protein fitness prediction under deliberately different generalization splits.","sections":[{"title":"Procedure","body":"Select an active released split, fit the selected sequence-to-fitness method on its training data and assess its test variants. The repository documents split construction and baseline implementations.","source_ids":["src-discovery-j-snackkb-flip"],"source_locator":"README: Folder breakup; Find out more about the splits; Split semaphore"}],"facts":[{"label":"Record type","value":"Dataset and split suite","source_ids":["src-discovery-j-snackkb-flip"],"source_locator":"README: Folder breakup; Find out more about the splits; Split semaphore"},{"label":"Inputs","value":"Protein sequences with experimental fitness measurements","source_ids":["src-discovery-j-snackkb-flip"],"source_locator":"README: Folder breakup; Find out more about the splits; Split semaphore"},{"label":"Outputs and assessment","value":"Fitness prediction on the selected split","source_ids":["src-discovery-j-snackkb-flip"],"source_locator":"README: Folder breakup; Find out more about the splits; Split semaphore"}],"strengths":[{"text":"Split definitions make the intended generalization challenge explicit.","source_ids":["src-discovery-j-snackkb-flip"],"source_locator":"README: Folder breakup; Find out more about the splits; Split semaphore"}],"limitations":[{"text":"Orange splits may overestimate performance; red splits are obsolete and should not support new comparisons.","source_ids":["src-discovery-j-snackkb-flip"],"source_locator":"README: Folder breakup; Find out more about the splits; Split semaphore"}],"diagram":{"title":"Procedure overview","steps":["Select active split","Fit fitness predictor","Predict held-out variants","Assess fitness predictions"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-j-snackkb-flip"],"source_locator":"README: Folder breakup; Find out more about the splits; Split semaphore"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Generalisation in protein fitness landscapes","facets":{"areas":["protein-function"]},"id":"discovery-benchmark-flip","kind":"benchmark","links":[],"name":"FLIP","source_ids":["src-discovery-j-snackkb-flip"],"status":"discovered"} {"attributes":{"entity_level":"suite","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Expanded protein fitness landscapes","version":null,"profile":{"summary":"FLIP2 extends protein fitness evaluation to additional engineering settings.","sections":[{"title":"Procedure","body":"Use a released split testing mutation count, position, unseen mutations, fitness range or wild-type transfer. Compare zero-shot models, ridge regression and fine-tuned predictors under the same split.","source_ids":["src-discovery-flip2"],"source_locator":"Official FLIP2 site: Abstract; Key Features; Overview; Split Types"}],"facts":[{"label":"Record type","value":"Dataset and split suite","source_ids":["src-discovery-flip2"],"source_locator":"Official FLIP2 site: Abstract; Key Features; Overview; Split Types"},{"label":"Inputs","value":"Protein variant sequences and measured properties","source_ids":["src-discovery-flip2"],"source_locator":"Official FLIP2 site: Abstract; Key Features; Overview; Split Types"},{"label":"Outputs and assessment","value":"Fitness prediction under explicit distribution shifts","source_ids":["src-discovery-flip2"],"source_locator":"Official FLIP2 site: Abstract; Key Features; Overview; Split Types"}],"strengths":[{"text":"Includes simple regression baselines and several practical transfer settings.","source_ids":["src-discovery-flip2"],"source_locator":"Official FLIP2 site: Abstract; Key Features; Overview; Split Types"}],"limitations":[{"text":"Split difficulty changes the question being asked; aggregate rankings do not identify a universal protein engineering method.","source_ids":["src-discovery-flip2"],"source_locator":"Official FLIP2 site: Abstract; Key Features; Overview; Split Types"}],"diagram":{"title":"Procedure overview","steps":["Choose engineering shift","Fit allowed predictor","Predict held-out variants","Compare measured fitness"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-flip2"],"source_locator":"Official FLIP2 site: Abstract; Key Features; Overview; Split Types"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Expanded protein fitness landscapes","facets":{"areas":["protein-function"]},"id":"discovery-benchmark-flip2","kind":"benchmark","links":[],"name":"FLIP2","source_ids":["src-discovery-flip2"],"status":"discovered"} {"attributes":{"entity_level":"suite","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Frozen genomic representations with linear probes","version":null,"profile":{"summary":"GENEB compares frozen DNA representations using a shared linear probe.","sections":[{"title":"Procedure","body":"Encode each DNA sequence, pool its hidden states and fit logistic regression without fine-tuning the encoder. Evaluate released train/test partitions in full-data, ten-shot and one-shot settings using the specified repeated seeds.","source_ids":["src-discovery-darlednik-geneb"],"source_locator":"README: Benchmark; Evaluation protocol; Evaluate your model"}],"facts":[{"label":"Record type","value":"Frozen-representation benchmark suite","source_ids":["src-discovery-darlednik-geneb"],"source_locator":"README: Benchmark; Evaluation protocol; Evaluate your model"},{"label":"Inputs","value":"DNA classification tasks and frozen sequence embeddings","source_ids":["src-discovery-darlednik-geneb"],"source_locator":"README: Benchmark; Evaluation protocol; Evaluate your model"},{"label":"Outputs and assessment","value":"MCC, accuracy and macro-F1 by task and category","source_ids":["src-discovery-darlednik-geneb"],"source_locator":"README: Benchmark; Evaluation protocol; Evaluate your model"}],"strengths":[{"text":"A shared probing protocol separates representation quality from model-specific fine-tuning.","source_ids":["src-discovery-darlednik-geneb"],"source_locator":"README: Benchmark; Evaluation protocol; Evaluate your model"}],"limitations":[{"text":"This tests frozen embeddings, not the best possible fine-tuned pipeline; full-data and few-shot rankings may differ.","source_ids":["src-discovery-darlednik-geneb"],"source_locator":"README: Benchmark; Evaluation protocol; Evaluate your model"}],"diagram":{"title":"Procedure overview","steps":["Released DNA splits","Frozen embeddings","Logistic-regression probe","Task and category metrics"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-darlednik-geneb"],"source_locator":"README: Benchmark; Evaluation protocol; Evaluate your model"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Frozen genomic representations with linear probes","facets":{"areas":["genomics"]},"id":"discovery-benchmark-geneb","kind":"benchmark","links":[],"name":"GENEB","source_ids":["src-discovery-darlednik-geneb"],"status":"discovered"} {"attributes":{"entity_level":"suite","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Genomic sequence classification","version":null,"profile":{"summary":"Genomic Benchmarks packages genomic sequence classification datasets.","sections":[{"title":"Procedure","body":"Download a named and versioned dataset, retain its released training and test folders, then train and assess a classifier. Dataset metadata describe class labels and sequence properties.","source_ids":["src-discovery-ml-bioinfo-ceitec-genomic-benchmarks"],"source_locator":"README: Usage / info and download_dataset; Structure of package"}],"facts":[{"label":"Record type","value":"Dataset collection and utilities","source_ids":["src-discovery-ml-bioinfo-ceitec-genomic-benchmarks"],"source_locator":"README: Usage / info and download_dataset; Structure of package"},{"label":"Inputs","value":"Versioned genomic sequence datasets","source_ids":["src-discovery-ml-bioinfo-ceitec-genomic-benchmarks"],"source_locator":"README: Usage / info and download_dataset; Structure of package"},{"label":"Outputs and assessment","value":"Sequence-classification predictions","source_ids":["src-discovery-ml-bioinfo-ceitec-genomic-benchmarks"],"source_locator":"README: Usage / info and download_dataset; Structure of package"}],"strengths":[{"text":"Standardized downloads and loaders lower barriers to repeatable data access.","source_ids":["src-discovery-ml-bioinfo-ceitec-genomic-benchmarks"],"source_locator":"README: Usage / info and download_dataset; Structure of package"}],"limitations":[{"text":"A dataset collection does not enforce one model-fitting or validation protocol; record those choices separately.","source_ids":["src-discovery-ml-bioinfo-ceitec-genomic-benchmarks"],"source_locator":"README: Usage / info and download_dataset; Structure of package"}],"diagram":{"title":"Procedure overview","steps":["Choose dataset version","Load train and test sequences","Train classifier","Score test predictions"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-ml-bioinfo-ceitec-genomic-benchmarks"],"source_locator":"README: Usage / info and download_dataset; Structure of package"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Genomic sequence classification","facets":{"areas":["genomics"]},"id":"discovery-benchmark-genomic-benchmarks","kind":"benchmark","links":[],"name":"Genomic Benchmarks","source_ids":["src-discovery-ml-bioinfo-ceitec-genomic-benchmarks"],"status":"discovered"} {"attributes":{"entity_level":"suite","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Glycan properties, taxonomy and molecular interactions","version":null,"profile":{"summary":"GlycanML evaluates glycan learning across taxonomy, immunogenicity, glycosylation and interaction tasks.","sections":[{"title":"Procedure","body":"Represent glycans as sequences or graphs, select a task configuration and train in either a single-task or multi-task setting. Use the corresponding released experiment configuration.","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"}],"facts":[{"label":"Record type","value":"Task suite","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"},{"label":"Inputs","value":"Glycan sequences or graphs and task labels","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"},{"label":"Outputs and assessment","value":"Task-specific glycan predictions","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"}],"strengths":[{"text":"Supports comparisons between sequence and graph representations.","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"}],"limitations":[{"text":"Single-task and multi-task training expose models to different supervision; the configuration must accompany a result.","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"}],"diagram":{"title":"Procedure overview","steps":["Select glycan task","Choose sequence or graph representation","Train configured model","Evaluate task"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Glycan properties, taxonomy and molecular interactions","facets":{"areas":["glycomics"]},"id":"discovery-benchmark-glycanml","kind":"benchmark","links":[],"name":"GlycanML","source_ids":["src-discovery-glycanml-glycanml"],"status":"discovered"} {"attributes":{"entity_level":"task","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"glycosylation type prediction","version":null,"profile":{"summary":"Predict the glycosylation category associated with a glycan.","sections":[{"title":"Procedure","body":"This is the GlycanML glycosylation type prediction component, not the parent suite as a whole. Represent glycans as sequences or graphs, select a task configuration and train in either a single-task or multi-task setting. Use the corresponding released experiment configuration.","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"}],"facts":[{"label":"Record type","value":"Component task","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"},{"label":"Inputs","value":"Glycan representations","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"},{"label":"Outputs and assessment","value":"Glycosylation labels","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"}],"strengths":[{"text":"Supports comparisons between sequence and graph representations.","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"}],"limitations":[{"text":"Single-task and multi-task training expose models to different supervision; the configuration must accompany a result.","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"}],"diagram":{"title":"Component evaluation overview","steps":["Glycan representations","GlycanML glycosylation type prediction","Glycosylation labels"],"caption":"Conceptual component-task overview. Exact data versions, split files and scorer settings must be taken from the source.","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"},"coverage":"limited","gaps":["Component-specific split manifest and complete scoring configuration have not yet been extracted."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Existing catalogue evidence and cited extraction inspected. The specific protocol gaps listed here remain unresolved; numerical source checks do not constitute complete methods review."}}},"description":"glycosylation type prediction","facets":{"areas":["glycomics"]},"id":"discovery-benchmark-glycanml-glycosylation-type-prediction","kind":"benchmark","links":[{"relation":"parent","target_id":"discovery-benchmark-glycanml"},{"relation":"part_of","target_id":"discovery-benchmark-glycanml"}],"name":"GlycanML glycosylation type prediction","source_ids":["src-discovery-glycanml-glycanml"],"status":"discovered"} {"attributes":{"entity_level":"task","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"immunogenicity prediction","version":null,"profile":{"summary":"Predict annotated glycan immunogenicity.","sections":[{"title":"Procedure","body":"This is the GlycanML immunogenicity prediction component, not the parent suite as a whole. Represent glycans as sequences or graphs, select a task configuration and train in either a single-task or multi-task setting. Use the corresponding released experiment configuration.","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"}],"facts":[{"label":"Record type","value":"Component task","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"},{"label":"Inputs","value":"Glycan representations","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"},{"label":"Outputs and assessment","value":"Immunogenicity labels","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"}],"strengths":[{"text":"Supports comparisons between sequence and graph representations.","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"}],"limitations":[{"text":"Single-task and multi-task training expose models to different supervision; the configuration must accompany a result.","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"}],"diagram":{"title":"Component evaluation overview","steps":["Glycan representations","GlycanML immunogenicity prediction","Immunogenicity labels"],"caption":"Conceptual component-task overview. Exact data versions, split files and scorer settings must be taken from the source.","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"},"coverage":"limited","gaps":["Component-specific split manifest and complete scoring configuration have not yet been extracted."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Existing catalogue evidence and cited extraction inspected. The specific protocol gaps listed here remain unresolved; numerical source checks do not constitute complete methods review."}}},"description":"immunogenicity prediction","facets":{"areas":["glycomics"]},"id":"discovery-benchmark-glycanml-immunogenicity-prediction","kind":"benchmark","links":[{"relation":"parent","target_id":"discovery-benchmark-glycanml"},{"relation":"part_of","target_id":"discovery-benchmark-glycanml"}],"name":"GlycanML immunogenicity prediction","source_ids":["src-discovery-glycanml-glycanml"],"status":"discovered"} {"attributes":{"entity_level":"task","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"protein-glycan interaction prediction","version":null,"profile":{"summary":"Predict protein–glycan interaction labels.","sections":[{"title":"Procedure","body":"This is the GlycanML protein-glycan interaction prediction component, not the parent suite as a whole. Represent glycans as sequences or graphs, select a task configuration and train in either a single-task or multi-task setting. Use the corresponding released experiment configuration.","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"}],"facts":[{"label":"Record type","value":"Component task","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"},{"label":"Inputs","value":"Protein and glycan information","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"},{"label":"Outputs and assessment","value":"Interaction predictions","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"}],"strengths":[{"text":"Supports comparisons between sequence and graph representations.","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"}],"limitations":[{"text":"Single-task and multi-task training expose models to different supervision; the configuration must accompany a result.","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"}],"diagram":{"title":"Component evaluation overview","steps":["Protein and glycan information","GlycanML protein-glycan interaction prediction","Interaction predictions"],"caption":"Conceptual component-task overview. Exact data versions, split files and scorer settings must be taken from the source.","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"},"coverage":"limited","gaps":["Component-specific split manifest and complete scoring configuration have not yet been extracted."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Existing catalogue evidence and cited extraction inspected. The specific protocol gaps listed here remain unresolved; numerical source checks do not constitute complete methods review."}}},"description":"protein-glycan interaction prediction","facets":{"areas":["glycomics"]},"id":"discovery-benchmark-glycanml-protein-glycan-interaction-prediction","kind":"benchmark","links":[{"relation":"parent","target_id":"discovery-benchmark-glycanml"},{"relation":"part_of","target_id":"discovery-benchmark-glycanml"}],"name":"GlycanML protein-glycan interaction prediction","source_ids":["src-discovery-glycanml-glycanml"],"status":"discovered"} {"attributes":{"entity_level":"task","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"taxonomy prediction","version":null,"profile":{"summary":"Predict taxonomy labels associated with glycans.","sections":[{"title":"Procedure","body":"This is the GlycanML taxonomy prediction component, not the parent suite as a whole. Represent glycans as sequences or graphs, select a task configuration and train in either a single-task or multi-task setting. Use the corresponding released experiment configuration.","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"}],"facts":[{"label":"Record type","value":"Component task","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"},{"label":"Inputs","value":"Glycan representations","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"},{"label":"Outputs and assessment","value":"Taxonomy labels","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"}],"strengths":[{"text":"Supports comparisons between sequence and graph representations.","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"}],"limitations":[{"text":"Single-task and multi-task training expose models to different supervision; the configuration must accompany a result.","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"}],"diagram":{"title":"Component evaluation overview","steps":["Glycan representations","GlycanML taxonomy prediction","Taxonomy labels"],"caption":"Conceptual component-task overview. Exact data versions, split files and scorer settings must be taken from the source.","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"},"coverage":"limited","gaps":["Component-specific split manifest and complete scoring configuration have not yet been extracted."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Existing catalogue evidence and cited extraction inspected. The specific protocol gaps listed here remain unresolved; numerical source checks do not constitute complete methods review."}}},"description":"taxonomy prediction","facets":{"areas":["glycomics"]},"id":"discovery-benchmark-glycanml-taxonomy-prediction","kind":"benchmark","links":[{"relation":"parent","target_id":"discovery-benchmark-glycanml"},{"relation":"part_of","target_id":"discovery-benchmark-glycanml"}],"name":"GlycanML taxonomy prediction","source_ids":["src-discovery-glycanml-glycanml"],"status":"discovered"} {"attributes":{"entity_level":"suite","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Multi-species genome understanding tasks","version":null,"profile":{"summary":"GUE tests genome understanding across multiple datasets, tasks and species.","sections":[{"title":"Procedure","body":"Use the released GUE data and model-specific evaluation scripts. Record the particular dataset, fine-tuning setup and selected checkpoint for each comparison.","source_ids":["src-discovery-magics-lab-dnabert-2"],"source_locator":"README: 2.1 GUE: Genome Understanding Evaluation; 6.1 Evaluate models on GUE"}],"facts":[{"label":"Record type","value":"Task suite","source_ids":["src-discovery-magics-lab-dnabert-2"],"source_locator":"README: 2.1 GUE: Genome Understanding Evaluation; 6.1 Evaluate models on GUE"},{"label":"Inputs","value":"Genomic sequences and dataset-specific labels","source_ids":["src-discovery-magics-lab-dnabert-2"],"source_locator":"README: 2.1 GUE: Genome Understanding Evaluation; 6.1 Evaluate models on GUE"},{"label":"Outputs and assessment","value":"Genome task classification results","source_ids":["src-discovery-magics-lab-dnabert-2"],"source_locator":"README: 2.1 GUE: Genome Understanding Evaluation; 6.1 Evaluate models on GUE"}],"strengths":[{"text":"Provides evaluation scripts for several genomic model families.","source_ids":["src-discovery-magics-lab-dnabert-2"],"source_locator":"README: 2.1 GUE: Genome Understanding Evaluation; 6.1 Evaluate models on GUE"}],"limitations":[{"text":"The suite-level name does not establish that all model runs used identical checkpoint selection or adaptation.","source_ids":["src-discovery-magics-lab-dnabert-2"],"source_locator":"README: 2.1 GUE: Genome Understanding Evaluation; 6.1 Evaluate models on GUE"}],"diagram":{"title":"Procedure overview","steps":["Choose GUE dataset","Run model-specific fine-tuning","Select checkpoint","Evaluate held-out data"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-magics-lab-dnabert-2"],"source_locator":"README: 2.1 GUE: Genome Understanding Evaluation; 6.1 Evaluate models on GUE"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Multi-species genome understanding tasks","facets":{"areas":["genomics"]},"id":"discovery-benchmark-gue","kind":"benchmark","links":[],"name":"GUE","source_ids":["src-discovery-magics-lab-dnabert-2"],"status":"discovered"} {"attributes":{"entity_level":"suite","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Gene expression prediction from matched histology","version":null,"profile":{"summary":"HEST-Benchmark tests gene-expression prediction from matched histology.","sections":[{"title":"Procedure","body":"Encode spatially matched image regions and use the benchmark procedure to predict measured gene expression. This molecular prediction task is distinct from generic medical-image classification.","source_ids":["src-discovery-mahmoodlab-hest"],"source_locator":"README: What does this repository provide?; HEST-Benchmark; Benchmarking your own model"}],"facts":[{"label":"Record type","value":"Spatial transcriptomics benchmark","source_ids":["src-discovery-mahmoodlab-hest"],"source_locator":"README: What does this repository provide?; HEST-Benchmark; Benchmarking your own model"},{"label":"Inputs","value":"Histology regions paired with spatial gene-expression measurements","source_ids":["src-discovery-mahmoodlab-hest"],"source_locator":"README: What does this repository provide?; HEST-Benchmark; Benchmarking your own model"},{"label":"Outputs and assessment","value":"Prediction of selected gene-expression targets","source_ids":["src-discovery-mahmoodlab-hest"],"source_locator":"README: What does this repository provide?; HEST-Benchmark; Benchmarking your own model"}],"strengths":[{"text":"Paired morphology and transcriptomics provide a measurable molecular endpoint.","source_ids":["src-discovery-mahmoodlab-hest"],"source_locator":"README: What does this repository provide?; HEST-Benchmark; Benchmarking your own model"}],"limitations":[{"text":"A correlation with expression does not establish causal regulation; tissue, assay and split definitions remain essential.","source_ids":["src-discovery-mahmoodlab-hest"],"source_locator":"README: What does this repository provide?; HEST-Benchmark; Benchmarking your own model"}],"diagram":{"title":"Procedure overview","steps":["Matched histology regions","Image representation","Expression prediction","Compare spatial measurements"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-mahmoodlab-hest"],"source_locator":"README: What does this repository provide?; HEST-Benchmark; Benchmarking your own model"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Gene expression prediction from matched histology","facets":{"areas":["spatial-omics"]},"id":"discovery-benchmark-hest-benchmark","kind":"benchmark","links":[],"name":"HEST-Benchmark","source_ids":["src-discovery-mahmoodlab-hest"],"status":"discovered"} {"attributes":{"entity_level":"suite","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Molecular identification from tandem mass spectra","version":null,"profile":{"summary":"MassSpecGym evaluates molecular identification and discovery from tandem mass spectra.","sections":[{"title":"Procedure","body":"Choose de novo generation, candidate retrieval or spectrum simulation. Each challenge specifies its input information, splits and output scoring; formula-assisted settings are separate variants.","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"}],"facts":[{"label":"Record type","value":"Molecular measurement task suite","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"},{"label":"Inputs","value":"MS/MS spectra or molecular structures, depending on task","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"},{"label":"Outputs and assessment","value":"Generated molecules, candidate rankings or simulated spectra","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"}],"strengths":[{"text":"Defines complementary tasks around the same measurement modality.","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"}],"limitations":[{"text":"Chemical-formula assistance changes available information; tasks and assisted variants cannot be collapsed into one score.","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"}],"diagram":{"title":"Procedure overview","steps":["Choose challenge and inputs","Apply released split","Generate task outputs","Score task predictions"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Molecular identification from tandem mass spectra","facets":{"areas":["metabolomics"]},"id":"discovery-benchmark-massspecgym","kind":"benchmark","links":[],"name":"MassSpecGym","source_ids":["src-discovery-pluskal-lab-massspecgym"],"status":"discovered"} {"attributes":{"entity_level":"task","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Generate molecular structures from tandem mass spectra","version":null,"profile":{"summary":"Generate candidate molecular structures from a tandem mass spectrum.","sections":[{"title":"Procedure","body":"This is the MassSpecGym De novo molecule generation component, not the parent suite as a whole. Choose de novo generation, candidate retrieval or spectrum simulation. Each challenge specifies its input information, splits and output scoring; formula-assisted settings are separate variants.","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"}],"facts":[{"label":"Record type","value":"Component task","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"},{"label":"Inputs","value":"MS/MS spectrum, optionally a supplied molecular formula","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"},{"label":"Outputs and assessment","value":"Generated molecular structures","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"}],"strengths":[{"text":"Defines complementary tasks around the same measurement modality.","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"}],"limitations":[{"text":"Chemical-formula assistance changes available information; tasks and assisted variants cannot be collapsed into one score.","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"}],"diagram":{"title":"Component evaluation overview","steps":["MS/MS spectrum, optionally a supplied molecular formula","MassSpecGym De novo molecule generation","Generated molecular structures"],"caption":"Conceptual component-task overview. Exact data versions, split files and scorer settings must be taken from the source.","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"},"coverage":"limited","gaps":["Component-specific split manifest and complete scoring configuration have not yet been extracted."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Existing catalogue evidence and cited extraction inspected. The specific protocol gaps listed here remain unresolved; numerical source checks do not constitute complete methods review."}}},"description":"Generate molecular structures from tandem mass spectra","facets":{"areas":["metabolomics"]},"id":"discovery-benchmark-massspecgym-de-novo-molecule-generation","kind":"benchmark","links":[{"relation":"parent","target_id":"discovery-benchmark-massspecgym"},{"relation":"part_of","target_id":"discovery-benchmark-massspecgym"}],"name":"MassSpecGym De novo molecule generation","source_ids":["src-discovery-pluskal-lab-massspecgym"],"status":"discovered"} {"attributes":{"entity_level":"task","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Rank candidate structures from a tandem mass spectrum","version":null,"profile":{"summary":"Rank candidate molecules for a tandem mass spectrum.","sections":[{"title":"Procedure","body":"This is the MassSpecGym Molecule retrieval component, not the parent suite as a whole. Choose de novo generation, candidate retrieval or spectrum simulation. Each challenge specifies its input information, splits and output scoring; formula-assisted settings are separate variants.","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"}],"facts":[{"label":"Record type","value":"Component task","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"},{"label":"Inputs","value":"MS/MS spectrum and a defined candidate set","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"},{"label":"Outputs and assessment","value":"Candidate ranking","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"}],"strengths":[{"text":"Defines complementary tasks around the same measurement modality.","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"}],"limitations":[{"text":"Chemical-formula assistance changes available information; tasks and assisted variants cannot be collapsed into one score.","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"}],"diagram":{"title":"Component evaluation overview","steps":["MS/MS spectrum and a defined candidate set","MassSpecGym Molecule retrieval","Candidate ranking"],"caption":"Conceptual component-task overview. Exact data versions, split files and scorer settings must be taken from the source.","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"},"coverage":"limited","gaps":["Component-specific split manifest and complete scoring configuration have not yet been extracted."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Existing catalogue evidence and cited extraction inspected. The specific protocol gaps listed here remain unresolved; numerical source checks do not constitute complete methods review."}}},"description":"Rank candidate structures from a tandem mass spectrum","facets":{"areas":["metabolomics"]},"id":"discovery-benchmark-massspecgym-molecule-retrieval","kind":"benchmark","links":[{"relation":"parent","target_id":"discovery-benchmark-massspecgym"},{"relation":"part_of","target_id":"discovery-benchmark-massspecgym"}],"name":"MassSpecGym Molecule retrieval","source_ids":["src-discovery-pluskal-lab-massspecgym"],"status":"discovered"} {"attributes":{"entity_level":"task","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Predict a tandem mass spectrum from molecular structure","version":null,"profile":{"summary":"Predict a tandem mass spectrum from molecular structure.","sections":[{"title":"Procedure","body":"This is the MassSpecGym Spectrum simulation component, not the parent suite as a whole. Choose de novo generation, candidate retrieval or spectrum simulation. Each challenge specifies its input information, splits and output scoring; formula-assisted settings are separate variants.","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"}],"facts":[{"label":"Record type","value":"Component task","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"},{"label":"Inputs","value":"Molecular structure","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"},{"label":"Outputs and assessment","value":"Simulated MS/MS spectrum","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"}],"strengths":[{"text":"Defines complementary tasks around the same measurement modality.","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"}],"limitations":[{"text":"Chemical-formula assistance changes available information; tasks and assisted variants cannot be collapsed into one score.","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"}],"diagram":{"title":"Component evaluation overview","steps":["Molecular structure","MassSpecGym Spectrum simulation","Simulated MS/MS spectrum"],"caption":"Conceptual component-task overview. Exact data versions, split files and scorer settings must be taken from the source.","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"},"coverage":"limited","gaps":["Component-specific split manifest and complete scoring configuration have not yet been extracted."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Existing catalogue evidence and cited extraction inspected. The specific protocol gaps listed here remain unresolved; numerical source checks do not constitute complete methods review."}}},"description":"Predict a tandem mass spectrum from molecular structure","facets":{"areas":["metabolomics"]},"id":"discovery-benchmark-massspecgym-spectrum-simulation","kind":"benchmark","links":[{"relation":"parent","target_id":"discovery-benchmark-massspecgym"},{"relation":"part_of","target_id":"discovery-benchmark-massspecgym"}],"name":"MassSpecGym Spectrum simulation","source_ids":["src-discovery-pluskal-lab-massspecgym"],"status":"discovered"} {"attributes":{"entity_level":"suite","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"mRNA embedding quality on downstream tasks","version":null,"profile":{"summary":"mRNABench assesses genomic model embeddings on mRNA-specific downstream tasks.","sections":[{"title":"Procedure","body":"Load a named dataset, generate model embeddings and fit a configured linear probe. The documented example uses a homology-aware splitter; retain the splitter and species settings with each evaluation.","source_ids":["src-discovery-morrislab-mrnabench"],"source_locator":"README: opening; Usage; Dataset Catalog"}],"facts":[{"label":"Record type","value":"Representation benchmark suite","source_ids":["src-discovery-morrislab-mrnabench"],"source_locator":"README: opening; Usage; Dataset Catalog"},{"label":"Inputs","value":"mRNA sequences, task labels and a chosen split","source_ids":["src-discovery-morrislab-mrnabench"],"source_locator":"README: opening; Usage; Dataset Catalog"},{"label":"Outputs and assessment","value":"Task-specific linear-probe results","source_ids":["src-discovery-morrislab-mrnabench"],"source_locator":"README: opening; Usage; Dataset Catalog"}],"strengths":[{"text":"Exposes datasets, embedding generation and split selection as explicit steps.","source_ids":["src-discovery-morrislab-mrnabench"],"source_locator":"README: opening; Usage; Dataset Catalog"}],"limitations":[{"text":"The example split is not a universal setting; task data, homology partition and random seed must be recorded.","source_ids":["src-discovery-morrislab-mrnabench"],"source_locator":"README: opening; Usage; Dataset Catalog"}],"diagram":{"title":"Procedure overview","steps":["Choose mRNA dataset","Generate embeddings","Build split and probe","Report held-out metrics"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-morrislab-mrnabench"],"source_locator":"README: opening; Usage; Dataset Catalog"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"mRNA embedding quality on downstream tasks","facets":{"areas":["rna"]},"id":"discovery-benchmark-mrnabench","kind":"benchmark","links":[],"name":"mRNABench","source_ids":["src-discovery-morrislab-mrnabench"],"status":"discovered"} {"attributes":{"entity_level":"suite","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"DNA and RNA fitness prediction","version":null,"profile":{"summary":"NABench evaluates nucleotide models against measured DNA and RNA fitness.","sections":[{"title":"Procedure","body":"Select a nucleic-acid assay and evaluation setting: zero-shot, few-shot, supervised or transfer learning. Compare predictions with the matched assay measurements using that setting's protocol.","source_ids":["src-discovery-mrzzmrzz-nabench"],"source_locator":"README: Overview; Leaderboard; Baseline Models; Resources"}],"facts":[{"label":"Record type","value":"Fitness benchmark suite","source_ids":["src-discovery-mrzzmrzz-nabench"],"source_locator":"README: Overview; Leaderboard; Baseline Models; Resources"},{"label":"Inputs","value":"DNA or RNA mutant sequences and assay measurements","source_ids":["src-discovery-mrzzmrzz-nabench"],"source_locator":"README: Overview; Leaderboard; Baseline Models; Resources"},{"label":"Outputs and assessment","value":"Fitness prediction across distinct adaptation regimes","source_ids":["src-discovery-mrzzmrzz-nabench"],"source_locator":"README: Overview; Leaderboard; Baseline Models; Resources"}],"strengths":[{"text":"Collects diverse high-throughput nucleotide assays under a common framework.","source_ids":["src-discovery-mrzzmrzz-nabench"],"source_locator":"README: Overview; Leaderboard; Baseline Models; Resources"}],"limitations":[{"text":"Different molecule families and training regimes remain different prediction problems; assay coverage must accompany aggregate scores.","source_ids":["src-discovery-mrzzmrzz-nabench"],"source_locator":"README: Overview; Leaderboard; Baseline Models; Resources"}],"diagram":{"title":"Procedure overview","steps":["Select assay and regime","Score nucleotide variants","Compare experimental fitness","Summarize by assay"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-mrzzmrzz-nabench"],"source_locator":"README: Overview; Leaderboard; Baseline Models; Resources"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"DNA and RNA fitness prediction","facets":{"areas":["rna"]},"id":"discovery-benchmark-nabench","kind":"benchmark","links":[],"name":"NABench","source_ids":["src-discovery-mrzzmrzz-nabench"],"status":"discovered"} {"attributes":{"entity_level":"evaluator","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Metagenomic taxonomic profile evaluation","version":null,"profile":{"summary":"OPAL evaluates predicted microbial taxon abundances.","sections":[{"title":"Procedure","body":"Supply predicted and gold-standard profiles in the documented taxonomic format. Evaluate presence and abundance agreement for matched samples and taxonomic ranks.","source_ids":["src-discovery-cami-challenge-opal"],"source_locator":"README: opening; Computed metrics; Inputs; Running opal.py"}],"facts":[{"label":"Record type","value":"Evaluator","source_ids":["src-discovery-cami-challenge-opal"],"source_locator":"README: opening; Computed metrics; Inputs; Running opal.py"},{"label":"Inputs","value":"Predicted and reference taxonomic abundance profiles","source_ids":["src-discovery-cami-challenge-opal"],"source_locator":"README: opening; Computed metrics; Inputs; Running opal.py"},{"label":"Outputs and assessment","value":"Taxonomic profile performance summaries","source_ids":["src-discovery-cami-challenge-opal"],"source_locator":"README: opening; Computed metrics; Inputs; Running opal.py"}],"strengths":[{"text":"A shared scorer supports side-by-side profiler assessment.","source_ids":["src-discovery-cami-challenge-opal"],"source_locator":"README: opening; Computed metrics; Inputs; Running opal.py"}],"limitations":[{"text":"Taxonomy identifiers and ranks must agree with the reference; profiling does not assign each read to a bin.","source_ids":["src-discovery-cami-challenge-opal"],"source_locator":"README: opening; Computed metrics; Inputs; Running opal.py"}],"diagram":{"title":"Procedure overview","steps":["Taxon abundance profiles","Align ranks and references","Compute profile metrics","Compare profilers"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-cami-challenge-opal"],"source_locator":"README: opening; Computed metrics; Inputs; Running opal.py"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Metagenomic taxonomic profile evaluation","facets":{"areas":["microbiome"]},"id":"discovery-benchmark-opal","kind":"benchmark","links":[],"name":"OPAL","source_ids":["src-discovery-cami-challenge-opal"],"status":"discovered"} {"attributes":{"entity_level":"suite","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Community single-cell analysis benchmarks","version":null,"profile":{"summary":"Open Problems is a community platform for single-cell analysis benchmarks.","sections":[{"title":"Procedure","body":"Choose a specific published task, dataset and evaluation workflow. The platform links benchmark results and datasets; it is not itself one fixed protocol.","source_ids":["src-discovery-openproblems-bio-openproblems"],"source_locator":"README: opening and links to benchmarks and datasets; official benchmark directory"}],"facts":[{"label":"Record type","value":"Benchmark platform","source_ids":["src-discovery-openproblems-bio-openproblems"],"source_locator":"README: opening and links to benchmarks and datasets; official benchmark directory"},{"label":"Inputs","value":"Task-specific single-cell data and predictions","source_ids":["src-discovery-openproblems-bio-openproblems"],"source_locator":"README: opening and links to benchmarks and datasets; official benchmark directory"},{"label":"Outputs and assessment","value":"Task-specific evaluation reports","source_ids":["src-discovery-openproblems-bio-openproblems"],"source_locator":"README: opening and links to benchmarks and datasets; official benchmark directory"}],"strengths":[{"text":"Provides a common location for community-maintained tasks and results.","source_ids":["src-discovery-openproblems-bio-openproblems"],"source_locator":"README: opening and links to benchmarks and datasets; official benchmark directory"}],"limitations":[{"text":"The pinned overview is intentionally brief; a concrete task version, metric and split must be extracted before comparing results.","source_ids":["src-discovery-openproblems-bio-openproblems"],"source_locator":"README: opening and links to benchmarks and datasets; official benchmark directory"}],"diagram":{"title":"Procedure overview","steps":["Select published task","Load task dataset","Run task workflow","Inspect evaluation"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-openproblems-bio-openproblems"],"source_locator":"README: opening and links to benchmarks and datasets; official benchmark directory"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Community single-cell analysis benchmarks","facets":{"areas":["single-cell"]},"id":"discovery-benchmark-open-problems","kind":"benchmark","links":[{"relation":"evaluates_task","target_id":"catalog-task-cell-batch-integration"}],"name":"Open Problems","source_ids":["src-discovery-openproblems-bio-openproblems"],"status":"discovered"} {"attributes":{"entity_level":"suite","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Predicting cellular perturbation responses","version":null,"profile":{"summary":"PerturBench standardizes cellular perturbation prediction workflows.","sections":[{"title":"Procedure","body":"Load a curated single-cell dataset and its defined split, produce predicted perturbed expression, then evaluate an explicitly chosen aggregation and metric pipeline. Some datasets use generated splits; others require released manual partitions.","source_ids":["src-discovery-altoslabs-perturbench"],"source_locator":"README: Downloading and Preparing Datasets / Data Splitting; Usage / Evaluator Class; Automated Evaluation"}],"facts":[{"label":"Record type","value":"Benchmark framework","source_ids":["src-discovery-altoslabs-perturbench"],"source_locator":"README: Downloading and Preparing Datasets / Data Splitting; Usage / Evaluator Class; Automated Evaluation"},{"label":"Inputs","value":"Single-cell expression, perturbation metadata and predicted responses","source_ids":["src-discovery-altoslabs-perturbench"],"source_locator":"README: Downloading and Preparing Datasets / Data Splitting; Usage / Evaluator Class; Automated Evaluation"},{"label":"Outputs and assessment","value":"Metrics on aggregated expression or response changes","source_ids":["src-discovery-altoslabs-perturbench"],"source_locator":"README: Downloading and Preparing Datasets / Data Splitting; Usage / Evaluator Class; Automated Evaluation"}],"strengths":[{"text":"Separates expression aggregation, distance or correlation metrics and optional ranking assessment.","source_ids":["src-discovery-altoslabs-perturbench"],"source_locator":"README: Downloading and Preparing Datasets / Data Splitting; Usage / Evaluator Class; Automated Evaluation"}],"limitations":[{"text":"Means, log-fold changes and other representations answer different questions; the exact aggregation and split must accompany the score.","source_ids":["src-discovery-altoslabs-perturbench"],"source_locator":"README: Downloading and Preparing Datasets / Data Splitting; Usage / Evaluator Class; Automated Evaluation"}],"diagram":{"title":"Procedure overview","steps":["Curated perturbation data","Select released split","Predict response","Aggregate and evaluate"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-altoslabs-perturbench"],"source_locator":"README: Downloading and Preparing Datasets / Data Splitting; Usage / Evaluator Class; Automated Evaluation"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Predicting cellular perturbation responses","facets":{"areas":["single-cell"]},"id":"discovery-benchmark-perturbench","kind":"benchmark","links":[],"name":"PerturBench","source_ids":["src-discovery-altoslabs-perturbench"],"status":"discovered"} {"attributes":{"entity_level":"suite","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Parameter estimation for biological dynamical models","version":null,"profile":{"summary":"The PEtab collection provides data-based parameter-estimation problems for mechanistic models.","sections":[{"title":"Procedure","body":"Choose a model problem together with its experimental measurements and parameter definitions. Fit the mathematical model using a specified solver and optimization configuration.","source_ids":["src-discovery-benchmarking-initiative-benchmark-models-petab"],"source_locator":"README: introduction; Overview problem table"}],"facts":[{"label":"Record type","value":"Mechanistic model problem collection","source_ids":["src-discovery-benchmarking-initiative-benchmark-models-petab"],"source_locator":"README: introduction; Overview problem table"},{"label":"Inputs","value":"Mathematical models, observations, conditions and parameters","source_ids":["src-discovery-benchmarking-initiative-benchmark-models-petab"],"source_locator":"README: introduction; Overview problem table"},{"label":"Outputs and assessment","value":"Parameter fits and problem-specific optimization assessments","source_ids":["src-discovery-benchmarking-initiative-benchmark-models-petab"],"source_locator":"README: introduction; Overview problem table"}],"strengths":[{"text":"Packages experimental data with the models to be fitted.","source_ids":["src-discovery-benchmarking-initiative-benchmark-models-petab"],"source_locator":"README: introduction; Overview problem table"}],"limitations":[{"text":"A problem definition does not standardize solver tolerances, starting points, optimization budgets or a universal score.","source_ids":["src-discovery-benchmarking-initiative-benchmark-models-petab"],"source_locator":"README: introduction; Overview problem table"}],"diagram":{"title":"Procedure overview","steps":["Select PEtab problem","Configure solver and parameters","Fit experimental observations","Assess fit and computation"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-benchmarking-initiative-benchmark-models-petab"],"source_locator":"README: introduction; Overview problem table"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Parameter estimation for biological dynamical models","facets":{"areas":["mechanistic-biology"]},"id":"discovery-benchmark-petab-benchmark-collection","kind":"benchmark","links":[],"name":"PEtab benchmark collection","source_ids":["src-discovery-benchmarking-initiative-benchmark-models-petab"],"status":"discovered"} {"attributes":{"entity_level":"suite","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Protein representation evaluation","version":null,"profile":{"summary":"PFMBench provides protein foundation model evaluation across downstream tasks.","sections":[{"title":"Procedure","body":"Select a task, dataset, model and tuning configuration in its Hydra-based workflow. Keep fine-tuned tasks separate from zero-shot scoring procedures.","source_ids":["src-discovery-biomap-research-pfmbench"],"source_locator":"README: Overview; Features; Project Structure; Quick Start"}],"facts":[{"label":"Record type","value":"Task suite","source_ids":["src-discovery-biomap-research-pfmbench"],"source_locator":"README: Overview; Features; Project Structure; Quick Start"},{"label":"Inputs","value":"Protein representations and task-specific data","source_ids":["src-discovery-biomap-research-pfmbench"],"source_locator":"README: Overview; Features; Project Structure; Quick Start"},{"label":"Outputs and assessment","value":"Structure, function and other protein task metrics","source_ids":["src-discovery-biomap-research-pfmbench"],"source_locator":"README: Overview; Features; Project Structure; Quick Start"}],"strengths":[{"text":"A modular workflow makes models and tuning configurations explicit.","source_ids":["src-discovery-biomap-research-pfmbench"],"source_locator":"README: Overview; Features; Project Structure; Quick Start"}],"limitations":[{"text":"Task-specific data releases and metric settings require separate pinning; the task count is not a comparable scientific score.","source_ids":["src-discovery-biomap-research-pfmbench"],"source_locator":"README: Overview; Features; Project Structure; Quick Start"}],"diagram":{"title":"Procedure overview","steps":["Select task configuration","Load model and data","Train or score","Evaluate task"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-biomap-research-pfmbench"],"source_locator":"README: Overview; Features; Project Structure; Quick Start"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Protein representation evaluation","facets":{"areas":["protein-function"]},"id":"discovery-benchmark-pfmbench","kind":"benchmark","links":[],"name":"PFMBench","source_ids":["src-discovery-biomap-research-pfmbench"],"status":"discovered"} {"attributes":{"entity_level":"suite","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Protein-ligand interaction evaluation","version":null,"profile":{"summary":"PLINDER combines protein–ligand interaction data with evaluation resources.","sections":[{"title":"Procedure","body":"Pin the dataset release and iteration, choose a train/validation/test partition and evaluate a docking method on the selected held-out subset. Test subsets distinguish ligand, pocket and protein novelty.","source_ids":["src-discovery-plinder-org-plinder"],"source_locator":"README: About; Plinder versions; Gold standard benchmark sets; Known bugs"}],"facts":[{"label":"Record type","value":"Dataset and evaluation resource","source_ids":["src-discovery-plinder-org-plinder"],"source_locator":"README: About; Plinder versions; Gold standard benchmark sets; Known bugs"},{"label":"Inputs","value":"Protein–ligand structures and annotated splits","source_ids":["src-discovery-plinder-org-plinder"],"source_locator":"README: About; Plinder versions; Gold standard benchmark sets; Known bugs"},{"label":"Outputs and assessment","value":"Docking evaluation on specified novelty strata","source_ids":["src-discovery-plinder-org-plinder"],"source_locator":"README: About; Plinder versions; Gold standard benchmark sets; Known bugs"}],"strengths":[{"text":"Similarity annotations and stratified test sets help make generalization explicit.","source_ids":["src-discovery-plinder-org-plinder"],"source_locator":"README: About; Plinder versions; Gold standard benchmark sets; Known bugs"}],"limitations":[{"text":"Dataset release-date corrections and split revisions can change membership; the two-part data version must be retained.","source_ids":["src-discovery-plinder-org-plinder"],"source_locator":"README: About; Plinder versions; Gold standard benchmark sets; Known bugs"}],"diagram":{"title":"Procedure overview","steps":["Pin data release","Choose novelty split","Predict complexes","Evaluate selected subset"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-plinder-org-plinder"],"source_locator":"README: About; Plinder versions; Gold standard benchmark sets; Known bugs"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Protein-ligand interaction evaluation","facets":{"areas":["molecular-interactions"]},"id":"discovery-benchmark-plinder","kind":"benchmark","links":[],"name":"PLINDER","source_ids":["src-discovery-plinder-org-plinder"],"status":"discovered"} {"attributes":{"entity_level":"evaluator","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Geometric and chemical plausibility of molecular poses","version":null,"profile":{"summary":"PoseBusters checks geometric and chemical plausibility of molecular poses.","sections":[{"title":"Procedure","body":"Provide a generated ligand pose, optionally with a receptor and reference ligand, and run the appropriate molecule, docking or redocking checks. These check types require different inputs.","source_ids":["src-discovery-maabuu-posebusters"],"source_locator":"README: Usage, CLI and PoseBusters configuration examples"}],"facts":[{"label":"Record type","value":"Evaluator","source_ids":["src-discovery-maabuu-posebusters"],"source_locator":"README: Usage, CLI and PoseBusters configuration examples"},{"label":"Inputs","value":"Ligand coordinates; receptor and reference ligand where required","source_ids":["src-discovery-maabuu-posebusters"],"source_locator":"README: Usage, CLI and PoseBusters configuration examples"},{"label":"Outputs and assessment","value":"Validity checks appropriate to the supplied configuration","source_ids":["src-discovery-maabuu-posebusters"],"source_locator":"README: Usage, CLI and PoseBusters configuration examples"}],"strengths":[{"text":"Adds molecular plausibility checks to positional assessment.","source_ids":["src-discovery-maabuu-posebusters"],"source_locator":"README: Usage, CLI and PoseBusters configuration examples"}],"limitations":[{"text":"Passing geometry checks does not demonstrate binding affinity, biological activity or correct ranking of compounds.","source_ids":["src-discovery-maabuu-posebusters"],"source_locator":"README: Usage, CLI and PoseBusters configuration examples"}],"diagram":{"title":"Procedure overview","steps":["Load molecular pose","Choose check configuration","Run validity checks","Inspect failures"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-maabuu-posebusters"],"source_locator":"README: Usage, CLI and PoseBusters configuration examples"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Geometric and chemical plausibility of molecular poses","facets":{"areas":["molecular-interactions"]},"id":"discovery-benchmark-posebusters","kind":"benchmark","links":[],"name":"PoseBusters","source_ids":["src-discovery-maabuu-posebusters"],"status":"discovered"} {"attributes":{"entity_level":"suite","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Protein prediction, design and dynamics evaluation","version":null,"profile":{"summary":"ProteinBench evaluates protein prediction, design and dynamics across several dimensions.","sections":[{"title":"Procedure","body":"Choose the relevant modality-to-modality task and examine its quality, novelty, diversity and robustness measures. Structure-conditioned design and backbone generation use different evaluation panels.","source_ids":["src-discovery-proteinbench"],"source_locator":"Official ProteinBench site: Abstract; inverse-folding and backbone-design table captions"}],"facts":[{"label":"Record type","value":"Evaluation framework","source_ids":["src-discovery-proteinbench"],"source_locator":"Official ProteinBench site: Abstract; inverse-folding and backbone-design table captions"},{"label":"Inputs","value":"Protein sequences, structures or task-specific design conditions","source_ids":["src-discovery-proteinbench"],"source_locator":"Official ProteinBench site: Abstract; inverse-folding and backbone-design table captions"},{"label":"Outputs and assessment","value":"Separate quality, novelty, diversity and robustness measures","source_ids":["src-discovery-proteinbench"],"source_locator":"Official ProteinBench site: Abstract; inverse-folding and backbone-design table captions"}],"strengths":[{"text":"Makes trade-offs visible beyond a single generation-quality score.","source_ids":["src-discovery-proteinbench"],"source_locator":"Official ProteinBench site: Abstract; inverse-folding and backbone-design table captions"}],"limitations":[{"text":"Novelty and diversity need to be interpreted alongside quality; a computed design metric is not experimental validation.","source_ids":["src-discovery-proteinbench"],"source_locator":"Official ProteinBench site: Abstract; inverse-folding and backbone-design table captions"}],"diagram":{"title":"Procedure overview","steps":["Choose protein task","Generate predictions or designs","Measure quality and diversity","Inspect trade-offs"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-proteinbench"],"source_locator":"Official ProteinBench site: Abstract; inverse-folding and backbone-design table captions"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Protein prediction, design and dynamics evaluation","facets":{"areas":["protein-structure"]},"id":"discovery-benchmark-proteinbench","kind":"benchmark","links":[],"name":"ProteinBench","source_ids":["src-discovery-proteinbench"],"status":"discovered"} {"attributes":{"entity_level":"suite","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Protein variant effect prediction","version":null,"profile":{"summary":"ProteinGym evaluates mutation-effect predictors against experimental protein assays.","sections":[{"title":"Procedure","body":"Score each variant in a pinned assay release, evaluate within assays and then apply the published protein and function-category aggregation. Keep substitutions, indels, zero-shot and supervised tracks separate.","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"README: Overview; Results; Resources; Usage and reproducibility"}],"facts":[{"label":"Record type","value":"Assay and evaluation suite","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"README: Overview; Results; Resources; Usage and reproducibility"},{"label":"Inputs","value":"Protein variants, assay measurements and permitted model inputs","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"README: Overview; Results; Resources; Usage and reproducibility"},{"label":"Outputs and assessment","value":"Spearman and other track-specific metrics","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"README: Overview; Results; Resources; Usage and reproducibility"}],"strengths":[{"text":"Provides assay-level results and aggregation intended to reduce repeated-protein bias.","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"README: Overview; Results; Resources; Usage and reproducibility"}],"limitations":[{"text":"DMS fitness is assay-specific; MSA- and structure-informed methods use different inputs from sequence-only methods.","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"README: Overview; Results; Resources; Usage and reproducibility"}],"diagram":{"title":"Procedure overview","steps":["Pin assay release","Score protein variants","Compute assay metrics","Aggregate by protocol"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"README: Overview; Results; Resources; Usage and reproducibility"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Protein variant effect prediction","facets":{"areas":["protein-function"]},"id":"discovery-benchmark-proteingym","kind":"benchmark","links":[{"relation":"evaluates_task","target_id":"catalog-task-proteingym-effects"}],"name":"ProteinGym","source_ids":["src-discovery-oatml-markslab-proteingym"],"status":"discovered"} {"attributes":{"entity_level":"suite","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Single-cell integration evaluation","version":null,"profile":{"summary":"scIB measures both batch removal and biological preservation in single-cell integration.","sections":[{"title":"Procedure","body":"Preprocess annotated single-cell data, run an integration method and evaluate its representation. Inspect biological-conservation metrics alongside batch-correction metrics; the package, reusable pipeline and published study are separate resources.","source_ids":["src-discovery-theislab-scib"],"source_locator":"README: Package scib; Metrics; Integration Tools; Resources"}],"facts":[{"label":"Record type","value":"Evaluator and associated benchmarking study","source_ids":["src-discovery-theislab-scib"],"source_locator":"README: Package scib; Metrics; Integration Tools; Resources"},{"label":"Inputs","value":"Single-cell expression or chromatin data with batch and biological annotations","source_ids":["src-discovery-theislab-scib"],"source_locator":"README: Package scib; Metrics; Integration Tools; Resources"},{"label":"Outputs and assessment","value":"Biological-conservation and batch-correction metric panels","source_ids":["src-discovery-theislab-scib"],"source_locator":"README: Package scib; Metrics; Integration Tools; Resources"}],"strengths":[{"text":"Makes the two competing integration objectives visible.","source_ids":["src-discovery-theislab-scib"],"source_locator":"README: Package scib; Metrics; Integration Tools; Resources"}],"limitations":[{"text":"A well-mixed embedding can erase real biological differences; label-dependent scores also depend on the reference annotations.","source_ids":["src-discovery-theislab-scib"],"source_locator":"README: Package scib; Metrics; Integration Tools; Resources"}],"diagram":{"title":"Procedure overview","steps":["Annotated cell data","Integrate batches","Measure biological conservation","Measure batch correction"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-theislab-scib"],"source_locator":"README: Package scib; Metrics; Integration Tools; Resources"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Single-cell integration evaluation","facets":{"areas":["single-cell"]},"id":"discovery-benchmark-scib","kind":"benchmark","links":[{"relation":"evaluates_task","target_id":"catalog-task-cell-batch-integration"}],"name":"scIB","source_ids":["src-discovery-theislab-scib"],"status":"discovered"} {"attributes":{"entity_level":"evaluator","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Perturbation prediction metric calibration","version":null,"profile":{"summary":"scPertEval examines evaluation protocols for single-cell perturbation predictions.","sections":[{"title":"Procedure","body":"Specify the representation, metric, score transformation and reporting strategy. Use the software to score predictions or calibrate a protocol against its built-in controls.","source_ids":["src-discovery-virtual-cell-research-community-scperteval"],"source_locator":"README: introduction; Quick start"}],"facts":[{"label":"Record type","value":"Evaluation software and protocol calibration","source_ids":["src-discovery-virtual-cell-research-community-scperteval"],"source_locator":"README: introduction; Quick start"},{"label":"Inputs","value":"Predicted and observed perturbation responses","source_ids":["src-discovery-virtual-cell-research-community-scperteval"],"source_locator":"README: introduction; Quick start"},{"label":"Outputs and assessment","value":"Protocol-dependent scores and control calibration","source_ids":["src-discovery-virtual-cell-research-community-scperteval"],"source_locator":"README: introduction; Quick start"}],"strengths":[{"text":"Treats the choice of evaluation protocol as something to test explicitly.","source_ids":["src-discovery-virtual-cell-research-community-scperteval"],"source_locator":"README: introduction; Quick start"}],"limitations":[{"text":"A score is not interpretable without its representation and transformation; the calibration controls are not new biological measurements.","source_ids":["src-discovery-virtual-cell-research-community-scperteval"],"source_locator":"README: introduction; Quick start"}],"diagram":{"title":"Procedure overview","steps":["Choose protocol components","Score predictions","Evaluate controls","Report calibrated interpretation"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-virtual-cell-research-community-scperteval"],"source_locator":"README: introduction; Quick start"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Perturbation prediction metric calibration","facets":{"areas":["single-cell"]},"id":"discovery-benchmark-scperteval","kind":"benchmark","links":[],"name":"scPertEval","source_ids":["src-discovery-virtual-cell-research-community-scperteval"],"status":"discovered"} {"attributes":{"entity_level":"suite","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Protein representation learning tasks","version":null,"profile":{"summary":"TAPE evaluates protein embeddings on five supervised downstream tasks.","sections":[{"title":"Procedure","body":"Choose secondary structure, contact prediction, remote homology, fluorescence or stability. Train a task head and evaluate using the corresponding data and metric. The current PyTorch repository differs from the original publication code.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: introduction; List of Models and Tasks; Data; Leaderboard"}],"facts":[{"label":"Record type","value":"Task suite","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: introduction; List of Models and Tasks; Data; Leaderboard"},{"label":"Inputs","value":"Protein sequences and task-specific labels","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: introduction; List of Models and Tasks; Data; Leaderboard"},{"label":"Outputs and assessment","value":"Classification, contact precision or rank-correlation metrics","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: introduction; List of Models and Tasks; Data; Leaderboard"}],"strengths":[{"text":"Includes simple one-hot comparators alongside pretrained representations.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: introduction; List of Models and Tasks; Data; Leaderboard"}],"limitations":[{"text":"The maintainers explicitly advise using the original implementation to reproduce the original paper.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: introduction; List of Models and Tasks; Data; Leaderboard"}],"diagram":{"title":"Procedure overview","steps":["Select downstream task","Fit task predictor","Evaluate held-out proteins","Report task metric"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: introduction; List of Models and Tasks; Data; Leaderboard"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Protein representation learning tasks","facets":{"areas":["protein-function"]},"id":"discovery-benchmark-tape","kind":"benchmark","links":[],"name":"TAPE","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"entity_level":"task","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Contact Prediction","version":null,"profile":{"summary":"Predict residue contacts from protein sequence.","sections":[{"title":"Procedure","body":"This is the TAPE Contact Prediction component, not the parent suite as a whole. Choose secondary structure, contact prediction, remote homology, fluorescence or stability. Train a task head and evaluate using the corresponding data and metric. The current PyTorch repository differs from the original publication code.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Contact Prediction"}],"facts":[{"label":"Record type","value":"Component task","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Contact Prediction"},{"label":"Inputs","value":"ProteinNet sequence and structural-contact labels","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Contact Prediction"},{"label":"Outputs and assessment","value":"Precision among the top L/5 medium- and long-range contacts","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Contact Prediction"}],"strengths":[{"text":"Includes simple one-hot comparators alongside pretrained representations.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Contact Prediction"}],"limitations":[{"text":"The maintainers explicitly advise using the original implementation to reproduce the original paper.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Contact Prediction"}],"diagram":{"title":"Component evaluation overview","steps":["ProteinNet sequence and structural-contact labels","TAPE Contact Prediction","Precision among the top L/5 medium- and long-range contacts"],"caption":"Conceptual component-task overview. Exact data versions, split files and scorer settings must be taken from the source.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Contact Prediction"},"coverage":"limited","gaps":["Component-specific split manifest and complete scoring configuration have not yet been extracted."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Existing catalogue evidence and cited extraction inspected. The specific protocol gaps listed here remain unresolved; numerical source checks do not constitute complete methods review."}}},"description":"Contact Prediction","facets":{"areas":["protein-function"]},"id":"discovery-benchmark-tape-contact-prediction","kind":"benchmark","links":[{"relation":"parent","target_id":"discovery-benchmark-tape"},{"relation":"part_of","target_id":"discovery-benchmark-tape"}],"name":"TAPE Contact Prediction","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"entity_level":"task","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Fluorescence","version":null,"profile":{"summary":"Predict measured fluorescence from protein sequence.","sections":[{"title":"Procedure","body":"This is the TAPE Fluorescence component, not the parent suite as a whole. Choose secondary structure, contact prediction, remote homology, fluorescence or stability. Train a task head and evaluate using the corresponding data and metric. The current PyTorch repository differs from the original publication code.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Fluorescence"}],"facts":[{"label":"Record type","value":"Component task","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Fluorescence"},{"label":"Inputs","value":"Protein variants and fluorescence labels","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Fluorescence"},{"label":"Outputs and assessment","value":"Spearman rank correlation","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Fluorescence"}],"strengths":[{"text":"Includes simple one-hot comparators alongside pretrained representations.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Fluorescence"}],"limitations":[{"text":"The maintainers explicitly advise using the original implementation to reproduce the original paper.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Fluorescence"}],"diagram":{"title":"Component evaluation overview","steps":["Protein variants and fluorescence labels","TAPE Fluorescence","Spearman rank correlation"],"caption":"Conceptual component-task overview. Exact data versions, split files and scorer settings must be taken from the source.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Fluorescence"},"coverage":"limited","gaps":["Component-specific split manifest and complete scoring configuration have not yet been extracted."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Existing catalogue evidence and cited extraction inspected. The specific protocol gaps listed here remain unresolved; numerical source checks do not constitute complete methods review."}}},"description":"Fluorescence","facets":{"areas":["protein-function"]},"id":"discovery-benchmark-tape-fluorescence","kind":"benchmark","links":[{"relation":"parent","target_id":"discovery-benchmark-tape"},{"relation":"part_of","target_id":"discovery-benchmark-tape"}],"name":"TAPE Fluorescence","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"entity_level":"task","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Remote Homology Detection","version":null,"profile":{"summary":"Classify remote protein homology.","sections":[{"title":"Procedure","body":"This is the TAPE Remote Homology Detection component, not the parent suite as a whole. Choose secondary structure, contact prediction, remote homology, fluorescence or stability. Train a task head and evaluate using the corresponding data and metric. The current PyTorch repository differs from the original publication code.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Remote Homology Detection"}],"facts":[{"label":"Record type","value":"Component task","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Remote Homology Detection"},{"label":"Inputs","value":"Protein sequences and homology labels","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Remote Homology Detection"},{"label":"Outputs and assessment","value":"Top-1 classification accuracy","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Remote Homology Detection"}],"strengths":[{"text":"Includes simple one-hot comparators alongside pretrained representations.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Remote Homology Detection"}],"limitations":[{"text":"The maintainers explicitly advise using the original implementation to reproduce the original paper.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Remote Homology Detection"}],"diagram":{"title":"Component evaluation overview","steps":["Protein sequences and homology labels","TAPE Remote Homology Detection","Top-1 classification accuracy"],"caption":"Conceptual component-task overview. Exact data versions, split files and scorer settings must be taken from the source.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Remote Homology Detection"},"coverage":"limited","gaps":["Component-specific split manifest and complete scoring configuration have not yet been extracted."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Existing catalogue evidence and cited extraction inspected. The specific protocol gaps listed here remain unresolved; numerical source checks do not constitute complete methods review."}}},"description":"Remote Homology Detection","facets":{"areas":["protein-function"]},"id":"discovery-benchmark-tape-remote-homology-detection","kind":"benchmark","links":[{"relation":"parent","target_id":"discovery-benchmark-tape"},{"relation":"part_of","target_id":"discovery-benchmark-tape"}],"name":"TAPE Remote Homology Detection","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"entity_level":"task","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Secondary Structure","version":null,"profile":{"summary":"Predict secondary-structure labels for protein residues.","sections":[{"title":"Procedure","body":"This is the TAPE Secondary Structure component, not the parent suite as a whole. Choose secondary structure, contact prediction, remote homology, fluorescence or stability. Train a task head and evaluate using the corresponding data and metric. The current PyTorch repository differs from the original publication code.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Secondary Structure"}],"facts":[{"label":"Record type","value":"Component task","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Secondary Structure"},{"label":"Inputs","value":"Protein sequences and residue labels","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Secondary Structure"},{"label":"Outputs and assessment","value":"Three-class accuracy","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Secondary Structure"}],"strengths":[{"text":"Includes simple one-hot comparators alongside pretrained representations.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Secondary Structure"}],"limitations":[{"text":"The maintainers explicitly advise using the original implementation to reproduce the original paper.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Secondary Structure"}],"diagram":{"title":"Component evaluation overview","steps":["Protein sequences and residue labels","TAPE Secondary Structure","Three-class accuracy"],"caption":"Conceptual component-task overview. Exact data versions, split files and scorer settings must be taken from the source.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Secondary Structure"},"coverage":"limited","gaps":["Component-specific split manifest and complete scoring configuration have not yet been extracted."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Existing catalogue evidence and cited extraction inspected. The specific protocol gaps listed here remain unresolved; numerical source checks do not constitute complete methods review."}}},"description":"Secondary Structure","facets":{"areas":["protein-function"]},"id":"discovery-benchmark-tape-secondary-structure","kind":"benchmark","links":[{"relation":"parent","target_id":"discovery-benchmark-tape"},{"relation":"part_of","target_id":"discovery-benchmark-tape"}],"name":"TAPE Secondary Structure","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"entity_level":"task","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Stability","version":null,"profile":{"summary":"Predict measured protein stability from sequence.","sections":[{"title":"Procedure","body":"This is the TAPE Stability component, not the parent suite as a whole. Choose secondary structure, contact prediction, remote homology, fluorescence or stability. Train a task head and evaluate using the corresponding data and metric. The current PyTorch repository differs from the original publication code.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Stability"}],"facts":[{"label":"Record type","value":"Component task","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Stability"},{"label":"Inputs","value":"Protein sequences and stability labels","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Stability"},{"label":"Outputs and assessment","value":"Spearman rank correlation","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Stability"}],"strengths":[{"text":"Includes simple one-hot comparators alongside pretrained representations.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Stability"}],"limitations":[{"text":"The maintainers explicitly advise using the original implementation to reproduce the original paper.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Stability"}],"diagram":{"title":"Component evaluation overview","steps":["Protein sequences and stability labels","TAPE Stability","Spearman rank correlation"],"caption":"Conceptual component-task overview. Exact data versions, split files and scorer settings must be taken from the source.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Stability"},"coverage":"limited","gaps":["Component-specific split manifest and complete scoring configuration have not yet been extracted."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Existing catalogue evidence and cited extraction inspected. The specific protocol gaps listed here remain unresolved; numerical source checks do not constitute complete methods review."}}},"description":"Stability","facets":{"areas":["protein-function"]},"id":"discovery-benchmark-tape-stability","kind":"benchmark","links":[{"relation":"parent","target_id":"discovery-benchmark-tape"},{"relation":"part_of","target_id":"discovery-benchmark-tape"}],"name":"TAPE Stability","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"entity_level":"suite","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Molecular binding, biochemical activity and related specialist tasks","version":null,"profile":{"summary":"TDC organizes datasets and evaluation tools for therapeutic machine learning.","sections":[{"title":"Procedure","body":"For rewire, select an in-scope molecular binding, activity or molecular-design task, then specify the dataset, split and evaluator. The broader TDC platform also includes tasks outside this catalogue's molecular remit.","source_ids":["src-discovery-mims-harvard-tdc"],"source_locator":"README: Unique Features; Design of TDC; Data Loaders"}],"facts":[{"label":"Record type","value":"Task platform, molecular subset","source_ids":["src-discovery-mims-harvard-tdc"],"source_locator":"README: Unique Features; Design of TDC; Data Loaders"},{"label":"Inputs","value":"Task-specific molecular or biochemical data","source_ids":["src-discovery-mims-harvard-tdc"],"source_locator":"README: Unique Features; Design of TDC; Data Loaders"},{"label":"Outputs and assessment","value":"Dataset-specific predictions and evaluation","source_ids":["src-discovery-mims-harvard-tdc"],"source_locator":"README: Unique Features; Design of TDC; Data Loaders"}],"strengths":[{"text":"Provides dataset loaders, split utilities and evaluation tools.","source_ids":["src-discovery-mims-harvard-tdc"],"source_locator":"README: Unique Features; Design of TDC; Data Loaders"}],"limitations":[{"text":"The full platform includes clinical tasks outside rewire's scope; inclusion must be decided per task, not by platform name.","source_ids":["src-discovery-mims-harvard-tdc"],"source_locator":"README: Unique Features; Design of TDC; Data Loaders"}],"diagram":{"title":"Procedure overview","steps":["Select molecular task","Pin dataset and split","Predict biochemical outcome","Evaluate chosen metric"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-mims-harvard-tdc"],"source_locator":"README: Unique Features; Design of TDC; Data Loaders"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Molecular binding, biochemical activity and related specialist tasks","facets":{"areas":["molecular-interactions"]},"id":"discovery-benchmark-tdc-molecular-tasks","kind":"benchmark","links":[],"name":"TDC molecular tasks","source_ids":["src-discovery-mims-harvard-tdc"],"status":"discovered"} {"attributes":{"entity_level":"challenge","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Zero-shot perturbation prediction in unseen cellular contexts","version":null,"profile":{"summary":"The 2026 Virtual Cell Challenge tests perturbation prediction in unseen cellular contexts.","sections":[{"title":"Procedure","body":"Predict post-CRISPRi expression from non-targeting-control cells and target-gene identifiers. No challenge-specific training responses are released. Three cell lines support validation and three are reserved for final testing.","source_ids":["src-discovery-vcc2026"],"source_locator":"Arc announcement, 20 August 2026: The 2026 task; Evaluation"}],"facts":[{"label":"Record type","value":"Prospective challenge","source_ids":["src-discovery-vcc2026"],"source_locator":"Arc announcement, 20 August 2026: The 2026 task; Evaluation"},{"label":"Inputs","value":"Unperturbed expression and CRISPRi target genes","source_ids":["src-discovery-vcc2026"],"source_locator":"Arc announcement, 20 August 2026: The 2026 task; Evaluation"},{"label":"Assessment","value":"Held-out expression responses, scored with the challenge metric panel","source_ids":["src-discovery-vcc2026"],"source_locator":"Arc announcement, 20 August 2026: The 2026 task; Evaluation"}],"strengths":[{"text":"Tests transfer between cellular contexts rather than interpolation among measured responses in one context.","source_ids":["src-discovery-vcc2026"],"source_locator":"Arc announcement, 20 August 2026: The 2026 task; Evaluation"}],"limitations":[{"text":"External training data are allowed; their provenance must be recorded. The final 2026 assessment is not yet available at this review date.","source_ids":["src-discovery-vcc2026"],"source_locator":"Arc announcement, 20 August 2026: The 2026 task; Evaluation"}],"diagram":{"title":"Procedure overview","steps":["Unperturbed cells and targets","Predict CRISPRi response","Withheld experimental response","Apply challenge scorer"],"caption":"Conceptual overview, not an executable specification.","source_ids":["src-discovery-vcc2026"],"source_locator":"Arc announcement, 20 August 2026: The 2026 task; Evaluation"},"coverage":"reviewed","gaps":["Pin the final cell-eval version, metric definitions and aggregation rules before interpreting final scores."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is profile review, not independent execution or numerical reproduction."}}},"description":"Zero-shot perturbation prediction in unseen cellular contexts","facets":{"areas":["single-cell"]},"id":"discovery-benchmark-virtual-cell-challenge-2026","kind":"benchmark","links":[],"name":"Virtual Cell Challenge 2026","source_ids":["src-discovery-vcc2026"],"status":"discovered"} {"attributes":{"assay":"mass spectrometry","missing_metadata":{"split":"not_applicable","version":"unextracted"},"scope_note":"Reference library; a leakage-aware benchmark split and scoring protocol must be defined separately.","split":null,"version":null},"description":"Experimental lipid reference spectra for identification assessment.","facets":{"areas":["lipidomics"]},"id":"discovery-dataset-lipid-maps-standards-spectra","kind":"dataset","links":[],"name":"LIPID MAPS Standards Spectra","source_ids":["src-discovery-lipidmaps-spectra"],"status":"discovered"} {"attributes":{"missing_metadata":{"denominator":"unextracted","split":"unextracted","version":"unreported"},"split":null,"version":null},"description":"Dataset used by the historical TAPE leaderboard.","facets":{"areas":["protein-function"]},"id":"discovery-dataset-tape-fluorescence-source-dataset","kind":"dataset","links":[],"name":"TAPE Fluorescence source dataset","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"missing_metadata":{"denominator":"unextracted","split":"unextracted","version":"unreported"},"split":null,"version":null},"description":"Dataset used by the historical TAPE leaderboard.","facets":{"areas":["protein-function"]},"id":"discovery-dataset-tape-stability-source-dataset","kind":"dataset","links":[],"name":"TAPE Stability source dataset","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"comparison":{"adaptation":null,"aggregation":null,"budget":null,"dataset_version":null,"inputs":null,"metric_implementation":null,"population":null,"protocol_id":"discovery-benchmark-tape-fluorescence","split":null},"missing_metadata":{"checkpoint":"unreported","denominator":"unextracted","seeds":"unreported","split_manifest":"unextracted"},"origin":"author_reported","protocol":"TAPE Fluorescence","version":"6d345c2b2bbf52cd32cf179325c222afd92aec7e"},"description":"Historical benchmark-maintainer report; experiments have not been rerun.","facets":{"areas":["protein-function"]},"id":"discovery-evaluation-tape-fluorescence-bepler-leaderboard-evaluation","kind":"evaluation","links":[{"relation":"model","target_id":"discovery-model-tape-bepler"},{"relation":"benchmark","target_id":"discovery-benchmark-tape-fluorescence"},{"relation":"dataset","target_id":"discovery-dataset-tape-fluorescence-source-dataset"}],"name":"TAPE Fluorescence Bepler leaderboard evaluation","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"comparison":{"adaptation":null,"aggregation":null,"budget":null,"dataset_version":null,"inputs":null,"metric_implementation":null,"population":null,"protocol_id":"discovery-benchmark-tape-fluorescence","split":null},"missing_metadata":{"checkpoint":"unreported","denominator":"unextracted","seeds":"unreported","split_manifest":"unextracted"},"origin":"author_reported","protocol":"TAPE Fluorescence","version":"6d345c2b2bbf52cd32cf179325c222afd92aec7e"},"description":"Historical benchmark-maintainer report; experiments have not been rerun.","facets":{"areas":["protein-function"]},"id":"discovery-evaluation-tape-fluorescence-lstm-leaderboard-evaluation","kind":"evaluation","links":[{"relation":"model","target_id":"discovery-model-tape-lstm"},{"relation":"benchmark","target_id":"discovery-benchmark-tape-fluorescence"},{"relation":"dataset","target_id":"discovery-dataset-tape-fluorescence-source-dataset"}],"name":"TAPE Fluorescence LSTM leaderboard evaluation","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"comparison":{"adaptation":null,"aggregation":null,"budget":null,"dataset_version":null,"inputs":null,"metric_implementation":null,"population":null,"protocol_id":"discovery-benchmark-tape-fluorescence","split":null},"missing_metadata":{"checkpoint":"unreported","denominator":"unextracted","seeds":"unreported","split_manifest":"unextracted"},"origin":"author_reported","protocol":"TAPE Fluorescence","version":"6d345c2b2bbf52cd32cf179325c222afd92aec7e"},"description":"Historical benchmark-maintainer report; experiments have not been rerun.","facets":{"areas":["protein-function"]},"id":"discovery-evaluation-tape-fluorescence-one-hot-leaderboard-evaluation","kind":"evaluation","links":[{"relation":"model","target_id":"discovery-model-tape-one-hot"},{"relation":"benchmark","target_id":"discovery-benchmark-tape-fluorescence"},{"relation":"dataset","target_id":"discovery-dataset-tape-fluorescence-source-dataset"}],"name":"TAPE Fluorescence One Hot leaderboard evaluation","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"comparison":{"adaptation":null,"aggregation":null,"budget":null,"dataset_version":null,"inputs":null,"metric_implementation":null,"population":null,"protocol_id":"discovery-benchmark-tape-fluorescence","split":null},"missing_metadata":{"checkpoint":"unreported","denominator":"unextracted","seeds":"unreported","split_manifest":"unextracted"},"origin":"author_reported","protocol":"TAPE Fluorescence","version":"6d345c2b2bbf52cd32cf179325c222afd92aec7e"},"description":"Historical benchmark-maintainer report; experiments have not been rerun.","facets":{"areas":["protein-function"]},"id":"discovery-evaluation-tape-fluorescence-resnet-leaderboard-evaluation","kind":"evaluation","links":[{"relation":"model","target_id":"discovery-model-tape-resnet"},{"relation":"benchmark","target_id":"discovery-benchmark-tape-fluorescence"},{"relation":"dataset","target_id":"discovery-dataset-tape-fluorescence-source-dataset"}],"name":"TAPE Fluorescence ResNet leaderboard evaluation","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"comparison":{"adaptation":null,"aggregation":null,"budget":null,"dataset_version":null,"inputs":null,"metric_implementation":null,"population":null,"protocol_id":"discovery-benchmark-tape-fluorescence","split":null},"missing_metadata":{"checkpoint":"unreported","denominator":"unextracted","seeds":"unreported","split_manifest":"unextracted"},"origin":"author_reported","protocol":"TAPE Fluorescence","version":"6d345c2b2bbf52cd32cf179325c222afd92aec7e"},"description":"Historical benchmark-maintainer report; experiments have not been rerun.","facets":{"areas":["protein-function"]},"id":"discovery-evaluation-tape-fluorescence-transformer-leaderboard-evaluation","kind":"evaluation","links":[{"relation":"model","target_id":"discovery-model-tape-transformer"},{"relation":"benchmark","target_id":"discovery-benchmark-tape-fluorescence"},{"relation":"dataset","target_id":"discovery-dataset-tape-fluorescence-source-dataset"}],"name":"TAPE Fluorescence Transformer leaderboard evaluation","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"comparison":{"adaptation":null,"aggregation":null,"budget":null,"dataset_version":null,"inputs":null,"metric_implementation":null,"population":null,"protocol_id":"discovery-benchmark-tape-fluorescence","split":null},"missing_metadata":{"checkpoint":"unreported","denominator":"unextracted","seeds":"unreported","split_manifest":"unextracted"},"origin":"author_reported","protocol":"TAPE Fluorescence","version":"6d345c2b2bbf52cd32cf179325c222afd92aec7e"},"description":"Historical benchmark-maintainer report; experiments have not been rerun.","facets":{"areas":["protein-function"]},"id":"discovery-evaluation-tape-fluorescence-unirep-leaderboard-evaluation","kind":"evaluation","links":[{"relation":"model","target_id":"discovery-model-tape-unirep"},{"relation":"benchmark","target_id":"discovery-benchmark-tape-fluorescence"},{"relation":"dataset","target_id":"discovery-dataset-tape-fluorescence-source-dataset"}],"name":"TAPE Fluorescence Unirep leaderboard evaluation","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"comparison":{"adaptation":null,"aggregation":null,"budget":null,"dataset_version":null,"inputs":null,"metric_implementation":null,"population":null,"protocol_id":"discovery-benchmark-tape-stability","split":null},"missing_metadata":{"checkpoint":"unreported","denominator":"unextracted","seeds":"unreported","split_manifest":"unextracted"},"origin":"author_reported","protocol":"TAPE Stability","version":"6d345c2b2bbf52cd32cf179325c222afd92aec7e"},"description":"Historical benchmark-maintainer report; experiments have not been rerun.","facets":{"areas":["protein-function"]},"id":"discovery-evaluation-tape-stability-bepler-leaderboard-evaluation","kind":"evaluation","links":[{"relation":"model","target_id":"discovery-model-tape-bepler"},{"relation":"benchmark","target_id":"discovery-benchmark-tape-stability"},{"relation":"dataset","target_id":"discovery-dataset-tape-stability-source-dataset"}],"name":"TAPE Stability Bepler leaderboard evaluation","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"comparison":{"adaptation":null,"aggregation":null,"budget":null,"dataset_version":null,"inputs":null,"metric_implementation":null,"population":null,"protocol_id":"discovery-benchmark-tape-stability","split":null},"missing_metadata":{"checkpoint":"unreported","denominator":"unextracted","seeds":"unreported","split_manifest":"unextracted"},"origin":"author_reported","protocol":"TAPE Stability","version":"6d345c2b2bbf52cd32cf179325c222afd92aec7e"},"description":"Historical benchmark-maintainer report; experiments have not been rerun.","facets":{"areas":["protein-function"]},"id":"discovery-evaluation-tape-stability-lstm-leaderboard-evaluation","kind":"evaluation","links":[{"relation":"model","target_id":"discovery-model-tape-lstm"},{"relation":"benchmark","target_id":"discovery-benchmark-tape-stability"},{"relation":"dataset","target_id":"discovery-dataset-tape-stability-source-dataset"}],"name":"TAPE Stability LSTM leaderboard evaluation","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"comparison":{"adaptation":null,"aggregation":null,"budget":null,"dataset_version":null,"inputs":null,"metric_implementation":null,"population":null,"protocol_id":"discovery-benchmark-tape-stability","split":null},"missing_metadata":{"checkpoint":"unreported","denominator":"unextracted","seeds":"unreported","split_manifest":"unextracted"},"origin":"author_reported","protocol":"TAPE Stability","version":"6d345c2b2bbf52cd32cf179325c222afd92aec7e"},"description":"Historical benchmark-maintainer report; experiments have not been rerun.","facets":{"areas":["protein-function"]},"id":"discovery-evaluation-tape-stability-one-hot-leaderboard-evaluation","kind":"evaluation","links":[{"relation":"model","target_id":"discovery-model-tape-one-hot"},{"relation":"benchmark","target_id":"discovery-benchmark-tape-stability"},{"relation":"dataset","target_id":"discovery-dataset-tape-stability-source-dataset"}],"name":"TAPE Stability One Hot leaderboard evaluation","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"comparison":{"adaptation":null,"aggregation":null,"budget":null,"dataset_version":null,"inputs":null,"metric_implementation":null,"population":null,"protocol_id":"discovery-benchmark-tape-stability","split":null},"missing_metadata":{"checkpoint":"unreported","denominator":"unextracted","seeds":"unreported","split_manifest":"unextracted"},"origin":"author_reported","protocol":"TAPE Stability","version":"6d345c2b2bbf52cd32cf179325c222afd92aec7e"},"description":"Historical benchmark-maintainer report; experiments have not been rerun.","facets":{"areas":["protein-function"]},"id":"discovery-evaluation-tape-stability-resnet-leaderboard-evaluation","kind":"evaluation","links":[{"relation":"model","target_id":"discovery-model-tape-resnet"},{"relation":"benchmark","target_id":"discovery-benchmark-tape-stability"},{"relation":"dataset","target_id":"discovery-dataset-tape-stability-source-dataset"}],"name":"TAPE Stability ResNet leaderboard evaluation","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"comparison":{"adaptation":null,"aggregation":null,"budget":null,"dataset_version":null,"inputs":null,"metric_implementation":null,"population":null,"protocol_id":"discovery-benchmark-tape-stability","split":null},"missing_metadata":{"checkpoint":"unreported","denominator":"unextracted","seeds":"unreported","split_manifest":"unextracted"},"origin":"author_reported","protocol":"TAPE Stability","version":"6d345c2b2bbf52cd32cf179325c222afd92aec7e"},"description":"Historical benchmark-maintainer report; experiments have not been rerun.","facets":{"areas":["protein-function"]},"id":"discovery-evaluation-tape-stability-transformer-leaderboard-evaluation","kind":"evaluation","links":[{"relation":"model","target_id":"discovery-model-tape-transformer"},{"relation":"benchmark","target_id":"discovery-benchmark-tape-stability"},{"relation":"dataset","target_id":"discovery-dataset-tape-stability-source-dataset"}],"name":"TAPE Stability Transformer leaderboard evaluation","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"comparison":{"adaptation":null,"aggregation":null,"budget":null,"dataset_version":null,"inputs":null,"metric_implementation":null,"population":null,"protocol_id":"discovery-benchmark-tape-stability","split":null},"missing_metadata":{"checkpoint":"unreported","denominator":"unextracted","seeds":"unreported","split_manifest":"unextracted"},"origin":"author_reported","protocol":"TAPE Stability","version":"6d345c2b2bbf52cd32cf179325c222afd92aec7e"},"description":"Historical benchmark-maintainer report; experiments have not been rerun.","facets":{"areas":["protein-function"]},"id":"discovery-evaluation-tape-stability-unirep-leaderboard-evaluation","kind":"evaluation","links":[{"relation":"model","target_id":"discovery-model-tape-unirep"},{"relation":"benchmark","target_id":"discovery-benchmark-tape-stability"},{"relation":"dataset","target_id":"discovery-dataset-tape-stability-source-dataset"}],"name":"TAPE Stability Unirep leaderboard evaluation","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-geneb"],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"Agro Nucleotide Transformer","version":null,"profile":{"summary":"Agro Nucleotide Transformer: plant genomic representation family","sections":[{"title":"Available evidence","body":"The discovery record links to instadeepai/nucleotide-transformer official source. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["src-discovery-instadeepai-nucleotide-transformer"],"source_locator":"README.md"}],"facts":[],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["src-discovery-instadeepai-nucleotide-transformer"],"source_locator":"README.md"}],"coverage":"limited","gaps":["The linked README.md has not yielded a reviewed architecture or procedure for the configuration 'Agro Nucleotide Transformer' in this release.","Unresolved registry fields: checkpoint (unextracted), code licence (unextracted), parameters (unextracted), training cutoff (unextracted), training data (unextracted), version (unextracted), weights licence (unextracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}},"description":"Plant genomic representation family","facets":{"areas":["genomics"]},"id":"discovery-model-agro-nucleotide-transformer","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-geneb"}],"name":"Agro Nucleotide Transformer","source_ids":["src-discovery-instadeepai-nucleotide-transformer"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-plinder"],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"AlphaFold 3","version":null,"profile":{"summary":"AlphaFold 3: biomolecular complex structure prediction","sections":[{"title":"Available evidence","body":"The discovery record links to google-deepmind/alphafold3 official source. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["src-discovery-google-deepmind-alphafold3"],"source_locator":"README.md"}],"facts":[],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["src-discovery-google-deepmind-alphafold3"],"source_locator":"README.md"}],"coverage":"limited","gaps":["The linked README.md has not yielded a reviewed architecture or procedure for the configuration 'AlphaFold 3' in this release.","Unresolved registry fields: checkpoint (unextracted), code licence (unextracted), parameters (unextracted), training cutoff (unextracted), training data (unextracted), version (unextracted), weights licence (unextracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}},"description":"Biomolecular complex structure prediction","facets":{"areas":["protein-structure"]},"id":"discovery-model-alphafold-3","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-plinder"}],"name":"AlphaFold 3","source_ids":["src-discovery-google-deepmind-alphafold3"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-posebusters"],"entity_level":"method","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"AutoDock Vina","version":null,"profile":{"summary":"AutoDock Vina is a conventional docking engine for searching ligand conformations against a receptor.","sections":[{"title":"How it works","body":"A scoring function guides gradient-based conformational search. Candidate poses are ranked under the selected docking configuration; receptor preparation and search settings are part of the method.","source_ids":["src-discovery-ccsb-scripps-autodock-vina"],"source_locator":"README.md: AutoDock Vina description and Features"}],"facts":[{"label":"Method class","value":"Docking search with a scoring function","source_ids":["src-discovery-ccsb-scripps-autodock-vina"],"source_locator":"README.md: AutoDock Vina description and Features"}],"strengths":[{"text":"Provides a procedural comparator with batch and multiple-ligand workflows.","source_ids":["src-discovery-ccsb-scripps-autodock-vina"],"source_locator":"README.md: AutoDock Vina description and Features"}],"limitations":[{"text":"A docking score is not an experimentally measured affinity. Pose and affinity tasks require separate evaluation.","source_ids":["src-discovery-ccsb-scripps-autodock-vina"],"source_locator":"README.md: AutoDock Vina description and Features"}],"diagram":{"title":"Conceptual procedure","steps":["Prepared receptor and ligand","Search configuration","Conformation search","Scoring function","Ranked docking poses"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-ccsb-scripps-autodock-vina"],"source_locator":"README.md: AutoDock Vina description and Features"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Procedural docking and virtual screening","facets":{"areas":["molecular-interactions"]},"id":"discovery-model-autodock-vina","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-posebusters"}],"name":"AutoDock Vina","source_ids":["src-discovery-ccsb-scripps-autodock-vina"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":[],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"Basenji","version":null,"profile":{"summary":"Basenji predicts regulatory activity along DNA with deep convolutional networks.","sections":[{"title":"How it works","body":"Sequence inputs are processed by convolutional models that output activity in bins. Variant scoring compares predicted activity for alternate sequences in a specified genomic context.","source_ids":["src-discovery-calico-basenji"],"source_locator":"README.md: opening goals and Changes versus Basset"}],"facts":[{"label":"Output form","value":"Regulatory activity in sequence bins","source_ids":["src-discovery-calico-basenji"],"source_locator":"README.md: opening goals and Changes versus Basset"}],"strengths":[{"text":"Supports sequence-to-activity and variant-scoring workflows.","source_ids":["src-discovery-calico-basenji"],"source_locator":"README.md: opening goals and Changes versus Basset"}],"limitations":[{"text":"Bin resolution, target tracks and sequence context affect what the prediction represents; a project-level record is not a specific trained regulatory model.","source_ids":["src-discovery-calico-basenji"],"source_locator":"README.md: opening goals and Changes versus Basset"}],"diagram":{"title":"Conceptual procedure","steps":["DNA sequence","Convolutional model","Binned regulatory activity","Allele comparison","Predicted regulatory effect"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-calico-basenji"],"source_locator":"README.md: opening goals and Changes versus Basset"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Sequence-to-regulatory-profile prediction","facets":{"areas":["genomics"]},"id":"discovery-model-basenji","kind":"model","links":[],"name":"Basenji","source_ids":["src-discovery-calico-basenji"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-plinder"],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"Boltz","version":null,"profile":{"summary":"Boltz is a biomolecular interaction model family. Boltz-2 adds affinity prediction to complex-structure prediction.","sections":[{"title":"Boltz-2 architecture","body":"The Boltz-2 implementation combines molecular and alignment features with a Pairformer module. A conditioned diffusion module predicts coordinates; a separate affinity module produces binding outputs. This architecture description applies to Boltz-2, not automatically to every Boltz family release.","source_ids":["src-discovery-jwohlwend-boltz"],"source_locator":"src/boltz/model/models/boltz2.py at the pinned repository revision: MSAModule, PairformerModule, DiffusionConditioning and AffinityModule; README Inference"},{"title":"How it works","body":"The documented YAML input describes the biomolecules and requested properties. Structure prediction and affinity outputs are distinct: one affinity output estimates binding strength, while another classifies binders against decoys.","source_ids":["src-discovery-jwohlwend-boltz"],"source_locator":"README.md: Introduction, Inference, Understanding the affinity prediction and License"}],"facts":[{"label":"Access","value":"Repository states code and models use the MIT licence","source_ids":["src-discovery-jwohlwend-boltz"],"source_locator":"README.md: Introduction, Inference, Understanding the affinity prediction and License"}],"strengths":[{"text":"Supports structure and affinity workflows in one openly distributed project.","source_ids":["src-discovery-jwohlwend-boltz"],"source_locator":"README.md: Introduction, Inference, Understanding the affinity prediction and License"}],"limitations":[{"text":"Binder probability and affinity regression are trained with different supervision and must not be compared as the same metric. Unqualified CLI calls select the latest model.","source_ids":["src-discovery-jwohlwend-boltz"],"source_locator":"README.md: Introduction, Inference, Understanding the affinity prediction and License"}],"diagram":{"title":"Conceptual procedure","steps":["Molecular inputs","MSA / Pairformer features","Coordinate diffusion","Structure","Separate affinity module"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-jwohlwend-boltz"],"source_locator":"Pinned repository src/boltz/model/models/boltz2.py, module construction and forward; README affinity prediction"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Biomolecular structure and affinity model family","facets":{"areas":["molecular-interactions"]},"id":"discovery-model-boltz","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-plinder"}],"name":"Boltz","source_ids":["src-discovery-jwohlwend-boltz"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-cami"],"entity_level":"method","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"CAMISIM","version":null,"profile":{"summary":"CAMISIM simulates shotgun metagenome datasets from modelled microbial-community abundances.","sections":[{"title":"How it works","body":"Community composition and sequencing configuration are used to generate synthetic metagenomic reads. The simulator can support benchmark construction but is not itself a taxonomic classifier.","source_ids":["src-discovery-cami-challenge-camisim"],"source_locator":"README.md: introduction and update to version 2.0"}],"facts":[{"label":"Record role","value":"Data-generation procedure","source_ids":["src-discovery-cami-challenge-camisim"],"source_locator":"README.md: introduction and update to version 2.0"}],"strengths":[{"text":"Produces controlled synthetic data for evaluating metagenomic workflows.","source_ids":["src-discovery-cami-challenge-camisim"],"source_locator":"README.md: introduction and update to version 2.0"}],"limitations":[{"text":"Simulation assumptions constrain realism. The documented CAMISIM2 and legacy 1.31-final workflows must be treated as distinct versions.","source_ids":["src-discovery-cami-challenge-camisim"],"source_locator":"README.md: introduction and update to version 2.0"}],"diagram":{"title":"Conceptual procedure","steps":["Community configuration","Abundance model","Read simulation","Synthetic metagenome","Benchmark input"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-cami-challenge-camisim"],"source_locator":"README.md: introduction and update to version 2.0"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Microbial community and metagenome simulation","facets":{"areas":["microbiome"]},"id":"discovery-model-camisim","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-cami"}],"name":"CAMISIM","source_ids":["src-discovery-cami-challenge-camisim"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-dart-eval"],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"ChromBPNet","version":null,"profile":{"summary":"ChromBPNet predicts chromatin-accessibility patterns at base resolution while separating assay bias from regulatory sequence signal.","sections":[{"title":"How it works","body":"The documented workflow trains bias-factorised sequence models for accessibility measurements. Predictions can be used for sequence interpretation and variant analysis, with assay and preprocessing settings retained.","source_ids":["src-discovery-kundajelab-chrombpnet"],"source_locator":"README.md: title, paper description and tutorial links"}],"facts":[{"label":"Output","value":"Base-resolution chromatin-accessibility predictions","source_ids":["src-discovery-kundajelab-chrombpnet"],"source_locator":"README.md: title, paper description and tutorial links"}],"strengths":[{"text":"Explicitly models assay bias when learning accessibility patterns.","source_ids":["src-discovery-kundajelab-chrombpnet"],"source_locator":"README.md: title, paper description and tutorial links"}],"limitations":[{"text":"A trained model is tied to its assay data and preprocessing; candidate applications are not verified benchmark results.","source_ids":["src-discovery-kundajelab-chrombpnet"],"source_locator":"README.md: title, paper description and tutorial links"}],"diagram":{"title":"Conceptual procedure","steps":["DNA and accessibility data","Bias-factorised model training","Sequence prediction","Base-resolution accessibility","Interpretation or variant analysis"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-kundajelab-chrombpnet"],"source_locator":"README.md: title, paper description and tutorial links"},"coverage":"reviewed","gaps":["This narrative covers the documented purpose; layer counts and checkpoint-specific training settings remain unextracted."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Bias-aware chromatin accessibility prediction","facets":{"areas":["genomics"]},"id":"discovery-model-chrombpnet","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-dart-eval"}],"name":"ChromBPNet","source_ids":["src-discovery-kundajelab-chrombpnet"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":[],"entity_level":"method","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"COBRApy","version":null,"profile":{"summary":"COBRApy provides constraint-based analyses of metabolic networks, rather than a pretrained neural checkpoint.","sections":[{"title":"How it works","body":"A metabolic reconstruction and constraints are passed to an optimisation solver. Analyses include flux balance, flux variability and gene-deletion experiments. Results depend on the reconstruction, objective, bounds and solver.","source_ids":["src-discovery-opencobra-cobrapy"],"source_locator":"README.rst: What is COBRApy?"}],"facts":[{"label":"Method class","value":"Constraint-based metabolic modelling","source_ids":["src-discovery-opencobra-cobrapy"],"source_locator":"README.rst: What is COBRApy?"}],"strengths":[{"text":"Provides explicit modelling assumptions and several mechanistic analysis procedures.","source_ids":["src-discovery-opencobra-cobrapy"],"source_locator":"README.rst: What is COBRApy?"}],"limitations":[{"text":"Software identity alone is insufficient to reproduce an analysis; biological constraints and solver settings must be recorded.","source_ids":["src-discovery-opencobra-cobrapy"],"source_locator":"README.rst: What is COBRApy?"}],"diagram":{"title":"Conceptual procedure","steps":["Metabolic reconstruction","Bounds and objective","Optimisation solver","Flux or deletion analysis","Predicted metabolic behaviour"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-opencobra-cobrapy"],"source_locator":"README.rst: What is COBRApy?"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Constraint-based metabolic modelling","facets":{"areas":["mechanistic-biology"]},"id":"discovery-model-cobrapy","kind":"model","links":[],"name":"COBRApy","source_ids":["src-discovery-opencobra-cobrapy"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-casp"],"entity_level":"method","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"ColabFold","version":null,"profile":{"summary":"ColabFold: protein folding pipeline","sections":[{"title":"Available evidence","body":"The discovery record links to sokrypton/ColabFold official source. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["src-discovery-sokrypton-colabfold"],"source_locator":"README.md"}],"facts":[],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["src-discovery-sokrypton-colabfold"],"source_locator":"README.md"}],"coverage":"limited","gaps":["The linked README.md has not yielded a reviewed architecture or procedure for the configuration 'ColabFold' in this release.","Unresolved registry fields: checkpoint (unextracted), code licence (unextracted), parameters (unextracted), training cutoff (unextracted), training data (unextracted), version (unextracted), weights licence (unextracted)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}},"description":"Protein folding pipeline","facets":{"areas":["protein-structure"]},"id":"discovery-model-colabfold","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-casp"}],"name":"ColabFold","source_ids":["src-discovery-sokrypton-colabfold"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-gue"],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"DNABERT-2","version":null,"profile":{"summary":"DNABERT-2 is a DNA encoder pretrained on sequences from multiple species. It supplies representations that can be adapted to genomic tasks.","sections":[{"title":"How it works","body":"DNA is compressed into variable-length byte-pair tokens. A BERT-style encoder uses ALiBi positional biases; masked-language pretraining learns contextual features. Sequence pooling or a separately trained prediction head produces task outputs.","source_ids":["src-discovery-magics-lab-dnabert-2"],"source_locator":"README.md: Introduction, Model and Data, Quick Start, Pre-Training and Fine-tune"}],"facts":[{"label":"Released model","value":"DNABERT-2-117M","source_ids":["src-discovery-magics-lab-dnabert-2"],"source_locator":"README.md: Introduction, Model and Data, Quick Start, Pre-Training and Fine-tune"},{"label":"Objective","value":"Masked-language pretraining","source_ids":["src-discovery-magics-lab-dnabert-2"],"source_locator":"README.md: Introduction, Model and Data, Quick Start, Pre-Training and Fine-tune"}],"strengths":[{"text":"One released encoder can support embedding extraction and supervised adaptation across several genomic tasks.","source_ids":["src-discovery-magics-lab-dnabert-2"],"source_locator":"README.md: Introduction, Model and Data, Quick Start, Pre-Training and Fine-tune"}],"limitations":[{"text":"An encoder embedding is not a splice-impact prediction. Pooling, sequence context and the supervised head are part of the evaluated pipeline.","source_ids":["src-discovery-magics-lab-dnabert-2"],"source_locator":"README.md: Introduction, Model and Data, Quick Start, Pre-Training and Fine-tune"}],"diagram":{"title":"Conceptual procedure","steps":["DNA sequence","Byte-pair tokens","BERT encoder with ALiBi","Token or pooled embeddings","Task-specific head"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-magics-lab-dnabert-2"],"source_locator":"README.md: Introduction, Model and Data, Quick Start, Pre-Training and Fine-tune"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Genomic sequence representation model","facets":{"areas":["genomics"]},"id":"discovery-model-dnabert-2","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-gue"}],"name":"DNABERT-2","source_ids":["src-discovery-magics-lab-dnabert-2"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-massspecgym"],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"DreaMS","version":null,"profile":{"summary":"DreaMS learns molecular representations from tandem mass spectra for downstream interpretation tasks.","sections":[{"title":"How it works","body":"A transformer is pretrained on unannotated spectra using masked spectral peaks and chromatographic retention ordering. Representations feed task-specific prediction or similarity workflows.","source_ids":["src-discovery-pluskal-lab-dreams"],"source_locator":"README.md: introduction and What can I do with DreaMS?"}],"facts":[{"label":"Training resource","value":"GeMS unannotated MS/MS spectra","source_ids":["src-discovery-pluskal-lab-dreams"],"source_locator":"README.md: introduction and What can I do with DreaMS?"}],"strengths":[{"text":"The project provides representations, spectra resources and downstream analysis workflows.","source_ids":["src-discovery-pluskal-lab-dreams"],"source_locator":"README.md: introduction and What can I do with DreaMS?"}],"limitations":[{"text":"An embedding or similarity score is not a definitive chemical identification; the candidate set and evaluation split remain essential.","source_ids":["src-discovery-pluskal-lab-dreams"],"source_locator":"README.md: introduction and What can I do with DreaMS?"}],"diagram":{"title":"Conceptual procedure","steps":["MS/MS spectrum","Spectral preprocessing","DreaMS transformer","Spectrum embedding","Task head or similarity"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-pluskal-lab-dreams"],"source_locator":"README.md: introduction and What can I do with DreaMS?"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Tandem mass spectrum representation model","facets":{"areas":["metabolomics"]},"id":"discovery-model-dreams","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-massspecgym"}],"name":"DreaMS","source_ids":["src-discovery-pluskal-lab-dreams"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-proteingym"],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"ESM-1v","version":null,"profile":{"summary":"ESM-1v: protein variant effect model family","sections":[{"title":"Available evidence","body":"The discovery record links to facebookresearch/esm official source. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md"}],"facts":[],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md"}],"coverage":"limited","gaps":["The linked README.md has not yielded a reviewed architecture or procedure for the configuration 'ESM-1v' in this release.","Unresolved registry fields: checkpoint (unextracted), code licence (unextracted), parameters (unextracted), training cutoff (unextracted), training data (unextracted), version (unextracted), weights licence (unextracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}},"description":"Protein variant effect model family","facets":{"areas":["protein-function"]},"id":"discovery-model-esm-1v","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-proteingym"}],"name":"ESM-1v","source_ids":["src-discovery-facebookresearch-esm"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-proteingym"],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"ESM-2","version":null,"profile":{"summary":"ESM-2 is a family of protein sequence transformers that produce residue-level and sequence-level representations.","sections":[{"title":"How it works","body":"Amino-acid tokens pass through a pretrained transformer. Hidden states can be retained for each residue or pooled for a whole protein; downstream tasks need an explicit scoring rule or predictor. ESMFold adds a structure-prediction system and is a separate pipeline.","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md: Main models, Getting started, Compute embeddings and Pre-trained Models"}],"facts":[{"label":"Training resource","value":"UniRef-derived protein sequences","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md: Main models, Getting started, Compute embeddings and Pre-trained Models"},{"label":"Configuration distinction","value":"The catalogue 8M entry is not the 650M or 15B checkpoint","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md: Main models, Getting started, Compute embeddings and Pre-trained Models"}],"strengths":[{"text":"Embeddings can be extracted directly from individual sequences; the repository provides several model sizes.","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md: Main models, Getting started, Compute embeddings and Pre-trained Models"}],"limitations":[{"text":"Different parameter sizes, pooling methods and supervised heads are not interchangeable evaluations.","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md: Main models, Getting started, Compute embeddings and Pre-trained Models"}],"diagram":{"title":"Conceptual procedure","steps":["Protein sequence","Amino-acid tokens","ESM-2 transformer","Residue embeddings","Pooling or task predictor"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md: Main models, Getting started, Compute embeddings and Pre-trained Models"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Protein sequence representation family","facets":{"areas":["protein-function"]},"id":"discovery-model-esm-2","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-proteingym"}],"name":"ESM-2","source_ids":["src-discovery-facebookresearch-esm"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-proteinbench"],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"ESM-IF1","version":null,"profile":{"summary":"ESM-IF1: protein inverse folding model","sections":[{"title":"Available evidence","body":"The discovery record links to facebookresearch/esm official source. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md"}],"facts":[],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md"}],"coverage":"limited","gaps":["The linked README.md has not yielded a reviewed architecture or procedure for the configuration 'ESM-IF1' in this release.","Unresolved registry fields: checkpoint (unextracted), code licence (unextracted), parameters (unextracted), training cutoff (unextracted), training data (unextracted), version (unextracted), weights licence (unextracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}},"description":"Protein inverse folding model","facets":{"areas":["protein-structure"]},"id":"discovery-model-esm-if1","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-proteinbench"}],"name":"ESM-IF1","source_ids":["src-discovery-facebookresearch-esm"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-proteinbench"],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"ESM3","version":null,"profile":{"summary":"ESM3: multimodal protein model family","sections":[{"title":"Available evidence","body":"The discovery record links to evolutionaryscale/esm official source. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["src-discovery-evolutionaryscale-esm"],"source_locator":"README.md"}],"facts":[],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["src-discovery-evolutionaryscale-esm"],"source_locator":"README.md"}],"coverage":"limited","gaps":["The linked README.md has not yielded a reviewed architecture or procedure for the configuration 'ESM3' in this release.","Unresolved registry fields: checkpoint (unextracted), code licence (unextracted), parameters (unextracted), training cutoff (unextracted), training data (unextracted), version (unextracted), weights licence (unextracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}},"description":"Multimodal protein model family","facets":{"areas":["protein-structure"]},"id":"discovery-model-esm3","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-proteinbench"}],"name":"ESM3","source_ids":["src-discovery-evolutionaryscale-esm"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":[],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"ESMC","version":null,"profile":{"summary":"ESMC: protein representation family","sections":[{"title":"Available evidence","body":"The discovery record links to evolutionaryscale/esm official source. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["src-discovery-evolutionaryscale-esm"],"source_locator":"README.md"}],"facts":[],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["src-discovery-evolutionaryscale-esm"],"source_locator":"README.md"}],"coverage":"limited","gaps":["The linked README.md has not yielded a reviewed architecture or procedure for the configuration 'ESMC' in this release.","Unresolved registry fields: checkpoint (unextracted), code licence (unextracted), parameters (unextracted), training cutoff (unextracted), training data (unextracted), version (unextracted), weights licence (unextracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}},"description":"Protein representation family","facets":{"areas":["protein-function"]},"id":"discovery-model-esmc","kind":"model","links":[],"name":"ESMC","source_ids":["src-discovery-evolutionaryscale-esm"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-proteinbench"],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"ESMFold","version":null,"profile":{"summary":"ESMFold predicts protein structures from individual amino-acid sequences using an ESM-2 representation model and a folding system.","sections":[{"title":"How it works","body":"The sequence is embedded and converted into a three-dimensional structure. The implementation exposes recycling and chunking controls; those choices affect memory use and the exact evaluated run.","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md: ESMFold Structure Prediction and Main models"}],"facts":[{"label":"Primary output","value":"PDB structure with confidence information","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md: ESMFold Structure Prediction and Main models"}],"strengths":[{"text":"The documented inference interface produces a structure directly from sequence.","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md: ESMFold Structure Prediction and Main models"}],"limitations":[{"text":"Long sequences and larger batches can exceed device memory. Version v0 and v1 refer to different released models.","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md: ESMFold Structure Prediction and Main models"}],"diagram":{"title":"Conceptual procedure","steps":["Protein sequence","ESM-2 features","Folding system","Recycling","Predicted structure"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md: ESMFold Structure Prediction and Main models"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Protein structure prediction model","facets":{"areas":["protein-structure"]},"id":"discovery-model-esmfold","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-proteinbench"}],"name":"ESMFold","source_ids":["src-discovery-facebookresearch-esm"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":[],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"ESMFold2","version":null,"profile":{"summary":"ESMFold2: protein structure and complex prediction","sections":[{"title":"Available evidence","body":"The discovery record links to evolutionaryscale/esm official source. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["src-discovery-evolutionaryscale-esm"],"source_locator":"README.md"}],"facts":[],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["src-discovery-evolutionaryscale-esm"],"source_locator":"README.md"}],"coverage":"limited","gaps":["The linked README.md has not yielded a reviewed architecture or procedure for the configuration 'ESMFold2' in this release.","Unresolved registry fields: checkpoint (unextracted), code licence (unextracted), parameters (unextracted), training cutoff (unextracted), training data (unextracted), version (unextracted), weights licence (unextracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}},"description":"Protein structure and complex prediction","facets":{"areas":["protein-structure"]},"id":"discovery-model-esmfold2","kind":"model","links":[],"name":"ESMFold2","source_ids":["src-discovery-evolutionaryscale-esm"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-geneb"],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"Evo 2","version":null,"profile":{"summary":"Evo 2 models and generates DNA at nucleotide resolution using the StripedHyena 2 architecture.","sections":[{"title":"How it works","body":"An autoregressive sequence model predicts successive nucleotides from preceding context. Scoring and generation use the selected released checkpoint; context length and device requirements depend on that configuration.","source_ids":["src-discovery-arcinstitute-evo2"],"source_locator":"README.md: introduction, Models and Installation"}],"facts":[{"label":"Training resource","value":"OpenGenome2","source_ids":["src-discovery-arcinstitute-evo2"],"source_locator":"README.md: introduction, Models and Installation"},{"label":"Objective","value":"Autoregressive sequence prediction","source_ids":["src-discovery-arcinstitute-evo2"],"source_locator":"README.md: introduction, Models and Installation"}],"strengths":[{"text":"The family is designed for long-context sequence modelling, with released inference code.","source_ids":["src-discovery-arcinstitute-evo2"],"source_locator":"README.md: introduction, Models and Installation"}],"limitations":[{"text":"A family-level maximum context length does not establish the settings used by a particular published evaluation.","source_ids":["src-discovery-arcinstitute-evo2"],"source_locator":"README.md: introduction, Models and Installation"}],"diagram":{"title":"Conceptual procedure","steps":["DNA nucleotides","StripedHyena 2","Autoregressive predictions","Sequence scoring or generation"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-arcinstitute-evo2"],"source_locator":"README.md: introduction, Models and Installation"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Genome sequence modelling family","facets":{"areas":["genomics"]},"id":"discovery-model-evo-2","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-geneb"}],"name":"Evo 2","source_ids":["src-discovery-arcinstitute-evo2"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-dart-eval"],"entity_level":"method","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"FIMO","version":null,"profile":{"summary":"FIMO searches sequences for matches to supplied motifs. It is a procedural motif-scanning comparator.","sections":[{"title":"How it works","body":"A motif set and sequence collection are supplied with an alphabet and background model. The background accounts for letter-frequency differences when scoring motif occurrences.","source_ids":["src-discovery-meme"],"source_locator":"FIMO submission documentation: Input motifs, Input sequences and Background model"}],"facts":[{"label":"Inputs","value":"Motifs, sequences and background model","source_ids":["src-discovery-meme"],"source_locator":"FIMO submission documentation: Input motifs, Input sequences and Background model"}],"strengths":[{"text":"Makes motif identity and background assumptions explicit.","source_ids":["src-discovery-meme"],"source_locator":"FIMO submission documentation: Input motifs, Input sequences and Background model"}],"limitations":[{"text":"Changing the background or motif collection changes the analysis. Motif occurrence is not by itself proof of binding or regulatory activity.","source_ids":["src-discovery-meme"],"source_locator":"FIMO submission documentation: Input motifs, Input sequences and Background model"}],"diagram":{"title":"Conceptual procedure","steps":["Motifs and sequences","Alphabet and background","Motif scanning","Candidate motif occurrences"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-meme"],"source_locator":"FIMO submission documentation: Input motifs, Input sequences and Background model"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Sequence motif scanning","facets":{"areas":["genomics"]},"id":"discovery-model-fimo","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-dart-eval"}],"name":"FIMO","source_ids":["src-discovery-meme"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-perturbench"],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"GEARS","version":null,"profile":{"summary":"GEARS predicts transcriptional responses to genetic perturbations using single-cell perturbation-screen data.","sections":[{"title":"How it works","body":"A task-specific model is trained on measured perturbations, then predicts gene-expression responses for requested single or combined perturbations. Training composition determines what generalisation question is being tested.","source_ids":["src-discovery-snap-stanford-gears"],"source_locator":"README.md: introduction and Important notes"}],"facts":[{"label":"Required evidence","value":"Perturbation identities and cells per condition","source_ids":["src-discovery-snap-stanford-gears"],"source_locator":"README.md: introduction and Important notes"}],"strengths":[{"text":"The implementation explicitly supports single-gene and multi-gene perturbation workflows.","source_ids":["src-discovery-snap-stanford-gears"],"source_locator":"README.md: introduction and Important notes"}],"limitations":[{"text":"The maintainers state that cross-cell-type transfer is unsupported and that reliable combinatorial prediction needs some combinatorial training data.","source_ids":["src-discovery-snap-stanford-gears"],"source_locator":"README.md: introduction and Important notes"}],"diagram":{"title":"Conceptual procedure","steps":["Perturbation-screen cells","Training perturbations","GEARS predictor","Requested perturbation","Expression response"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-snap-stanford-gears"],"source_locator":"README.md: introduction and Important notes"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Genetic perturbation response prediction","facets":{"areas":["single-cell"]},"id":"discovery-model-gears","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-perturbench"}],"name":"GEARS","source_ids":["src-discovery-snap-stanford-gears"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":[],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"Genie 3","version":null,"profile":{"summary":"Genie 3: equivariant all-atom protein design","sections":[{"title":"Available evidence","body":"The discovery record links to aqlaboratory/genie3 official source. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["src-discovery-aqlaboratory-genie3"],"source_locator":"README.md"}],"facts":[],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["src-discovery-aqlaboratory-genie3"],"source_locator":"README.md"}],"coverage":"limited","gaps":["The linked README.md has not yielded a reviewed architecture or procedure for the configuration 'Genie 3' in this release.","Unresolved registry fields: checkpoint (unextracted), code licence (unextracted), parameters (unextracted), training cutoff (unextracted), training data (unextracted), version (unextracted), weights licence (unextracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}},"description":"Equivariant all-atom protein design","facets":{"areas":["protein-structure"]},"id":"discovery-model-genie-3","kind":"model","links":[],"name":"Genie 3","source_ids":["src-discovery-aqlaboratory-genie3"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-beeline"],"entity_level":"method","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"GENIE3","version":null,"profile":{"summary":"GENIE3 infers candidate gene regulatory networks from gene-expression data using ensembles of trees.","sections":[{"title":"How it works","body":"Gene expression supplies the measurements used by the tree-ensemble inference method. Its output is a candidate network whose biological interpretation requires a specified evaluation protocol.","source_ids":["src-discovery-aertslab-genie3"],"source_locator":"README.md: GENIE3 introduction"}],"facts":[{"label":"Identity distinction","value":"GENIE3 network inference is separate from Genie 3 protein design","source_ids":["src-discovery-aertslab-genie3"],"source_locator":"README.md: GENIE3 introduction"}],"strengths":[{"text":"Provides an established non-foundation-model comparator for network inference.","source_ids":["src-discovery-aertslab-genie3"],"source_locator":"README.md: GENIE3 introduction"}],"limitations":[{"text":"Network prediction requires external reference edges or interventions for validation; the project name does not define a matched benchmark.","source_ids":["src-discovery-aertslab-genie3"],"source_locator":"README.md: GENIE3 introduction"}],"diagram":{"title":"Conceptual procedure","steps":["Gene-expression data","Tree-ensemble inference","Regulator relevance","Candidate network"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-aertslab-genie3"],"source_locator":"README.md: GENIE3 introduction"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Tree-ensemble gene regulatory network inference","facets":{"areas":["biological-networks"]},"id":"discovery-model-genie3","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-beeline"}],"name":"GENIE3","source_ids":["src-discovery-aertslab-genie3"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-glycanml"],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"GlycanGT","version":null,"profile":{"summary":"GlycanGT learns glycan representations with a graph transformer that treats both sugars and linkages as tokens.","sections":[{"title":"How it works","body":"Node and edge tokens include content, identifiers and token types. Transformer layers produce a graph-level embedding. Masked pretraining also supports prediction of missing glycan components.","source_ids":["src-discovery-matsui-lab-glycangt"],"source_locator":"README.md: Model architecture / Training details"}],"facts":[{"label":"Architecture","value":"TokenGT-based graph transformer","source_ids":["src-discovery-matsui-lab-glycangt"],"source_locator":"README.md: Model architecture / Training details"}],"strengths":[{"text":"Explicit linkage tokens allow the representation to include more than monosaccharide composition.","source_ids":["src-discovery-matsui-lab-glycangt"],"source_locator":"README.md: Model architecture / Training details"}],"limitations":[{"text":"The training collection excludes ambiguous symbols; completion predictions are hypotheses about missing structure, not experimental confirmation.","source_ids":["src-discovery-matsui-lab-glycangt"],"source_locator":"README.md: Model architecture / Training details"}],"diagram":{"title":"Conceptual procedure","steps":["Glycan graph","Sugar and linkage tokens","Graph transformer","Graph embedding","Classification or completion"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-matsui-lab-glycangt"],"source_locator":"README.md: Model architecture / Training details"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Glycan graph transformer","facets":{"areas":["glycomics"]},"id":"discovery-model-glycangt","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-glycanml"}],"name":"GlycanGT","source_ids":["src-discovery-matsui-lab-glycangt"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-beeline"],"entity_level":"method","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"GRNBoost","version":null,"profile":{"summary":"GRNBoost infers candidate regulatory networks by predicting gene expression with boosted trees.","sections":[{"title":"How it works","body":"For each target gene, expression from candidate regulators is used in a regression. Feature importance becomes evidence for candidate regulatory edges; the implementation distributes these regressions with Spark.","source_ids":["src-discovery-aertslab-grnboost"],"source_locator":"README.md: What is GRNBoost? and algorithm description"}],"facts":[{"label":"Implementation distinction","value":"This source describes GRNBoost; GRNBoost2 is not silently substituted","source_ids":["src-discovery-aertslab-grnboost"],"source_locator":"README.md: What is GRNBoost? and algorithm description"}],"strengths":[{"text":"Offers a scalable conventional learning comparator for network inference.","source_ids":["src-discovery-aertslab-grnboost"],"source_locator":"README.md: What is GRNBoost? and algorithm description"}],"limitations":[{"text":"Predictive importance in observational expression data is not proof of a direct causal regulatory edge.","source_ids":["src-discovery-aertslab-grnboost"],"source_locator":"README.md: What is GRNBoost? and algorithm description"}],"diagram":{"title":"Conceptual procedure","steps":["Expression matrix","Candidate regulators","Per-gene boosted regressions","Feature importance","Candidate regulatory network"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-aertslab-grnboost"],"source_locator":"README.md: What is GRNBoost? and algorithm description"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Boosted-tree regulatory network inference","facets":{"areas":["biological-networks"]},"id":"discovery-model-grnboost","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-beeline"}],"name":"GRNBoost","source_ids":["src-discovery-aertslab-grnboost"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-cafa"],"entity_level":"method","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"HH-suite","version":null,"profile":{"summary":"HH-suite searches for protein homologues by comparing sequence profiles represented as hidden Markov models.","sections":[{"title":"How it works","body":"A query profile is compared with profiles in a reference database. The resulting alignments and similarity evidence support homology-based analysis.","source_ids":["src-discovery-soedinglab-hh-suite"],"source_locator":"README.md: introduction"}],"facts":[{"label":"Method class","value":"Hidden Markov model profile alignment","source_ids":["src-discovery-soedinglab-hh-suite"],"source_locator":"README.md: introduction"}],"strengths":[{"text":"Provides a non-neural homology comparator using evolutionary profile information.","source_ids":["src-discovery-soedinglab-hh-suite"],"source_locator":"README.md: introduction"}],"limitations":[{"text":"Its input information includes a profile and database; it is not directly comparable to a single-sequence model without accounting for that extra information.","source_ids":["src-discovery-soedinglab-hh-suite"],"source_locator":"README.md: introduction"}],"diagram":{"title":"Conceptual procedure","steps":["Protein query profile","Reference profile database","HMM–HMM alignment","Ranked homologues"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-soedinglab-hh-suite"],"source_locator":"README.md: introduction"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Profile hidden Markov sequence search","facets":{"areas":["protein-function"]},"id":"discovery-model-hh-suite","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-cafa"}],"name":"HH-suite","source_ids":["src-discovery-soedinglab-hh-suite"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":[],"entity_level":"method","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"HUMAnN","version":null,"profile":{"summary":"HUMAnN: microbial functional profiling","sections":[{"title":"Available evidence","body":"The discovery record links to biobakery/humann official source. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["src-discovery-biobakery-humann"],"source_locator":"readme.md"}],"facts":[],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["src-discovery-biobakery-humann"],"source_locator":"readme.md"}],"coverage":"limited","gaps":["The linked readme.md has not yielded a reviewed architecture or procedure for the configuration 'HUMAnN' in this release.","Unresolved registry fields: checkpoint (unextracted), code licence (unextracted), parameters (unextracted), training cutoff (unextracted), training data (unextracted), version (unextracted), weights licence (unextracted)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}},"description":"Microbial functional profiling","facets":{"areas":["microbiome"]},"id":"discovery-model-humann","kind":"model","links":[],"name":"HUMAnN","source_ids":["src-discovery-biobakery-humann"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-cami"],"entity_level":"method","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"Kraken 2","version":null,"profile":{"summary":"Kraken 2: sequence-based metagenomic classification","sections":[{"title":"Available evidence","body":"The discovery record links to DerrickWood/kraken2 official source. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["src-discovery-derrickwood-kraken2"],"source_locator":"README.md"}],"facts":[],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["src-discovery-derrickwood-kraken2"],"source_locator":"README.md"}],"coverage":"limited","gaps":["The linked README.md has not yielded a reviewed architecture or procedure for the configuration 'Kraken 2' in this release.","Unresolved registry fields: checkpoint (unextracted), code licence (unextracted), parameters (unextracted), training cutoff (unextracted), training data (unextracted), version (unextracted), weights licence (unextracted)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}},"description":"Sequence-based metagenomic classification","facets":{"areas":["microbiome"]},"id":"discovery-model-kraken-2","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-cami"}],"name":"Kraken 2","source_ids":["src-discovery-derrickwood-kraken2"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":[],"entity_level":"method","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"LipidBlast","version":null,"profile":{"summary":"LipidBlast is an in-silico tandem mass spectral library used for lipid annotation by library search.","sections":[{"title":"How it works","body":"Experimentally acquired MS/MS spectra are compared with a computer-generated reference library. The library and search procedure together determine the candidate annotations.","source_ids":["src-discovery-lipidblast"],"source_locator":"Project page: LipidBlast in silico tandem mass spectrometry database for lipid identification; FAQ"}],"facts":[{"label":"Method class","value":"Generated spectral library, not a learned checkpoint","source_ids":["src-discovery-lipidblast"],"source_locator":"Project page: LipidBlast in silico tandem mass spectrometry database for lipid identification; FAQ"}],"strengths":[{"text":"Provides a procedural reference spanning multiple lipid classes and instrument types.","source_ids":["src-discovery-lipidblast"],"source_locator":"Project page: LipidBlast in silico tandem mass spectrometry database for lipid identification; FAQ"}],"limitations":[{"text":"Library matching depends on fragment information and acquisition conditions; a matching candidate does not resolve every structural isomer.","source_ids":["src-discovery-lipidblast"],"source_locator":"Project page: LipidBlast in silico tandem mass spectrometry database for lipid identification; FAQ"}],"diagram":{"title":"Conceptual procedure","steps":["Lipid MS/MS spectrum","Library search settings","In-silico reference spectra","Spectral matches","Candidate lipid annotations"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-lipidblast"],"source_locator":"Project page: LipidBlast in silico tandem mass spectrometry database for lipid identification; FAQ"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Rule-based lipid fragmentation library matching","facets":{"areas":["lipidomics"]},"id":"discovery-model-lipidblast","kind":"model","links":[],"name":"LipidBlast","source_ids":["src-discovery-lipidblast"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":[],"entity_level":"method","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"LipidFinder","version":null,"profile":{"summary":"LipidFinder: lC-MS lipid feature filtering and annotation","sections":[{"title":"Available evidence","body":"The discovery record links to lipidfinder official source. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["src-discovery-lipidfinder"],"source_locator":"HTML main page"}],"facts":[],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["src-discovery-lipidfinder"],"source_locator":"HTML main page"}],"coverage":"limited","gaps":["The linked HTML main page has not yielded a reviewed architecture or procedure for the configuration 'LipidFinder' in this release.","Unresolved registry fields: checkpoint (unextracted), code licence (unextracted), parameters (unextracted), training cutoff (unextracted), training data (unextracted), version (unextracted), weights licence (unextracted).","The linked project page returned HTTP 403 during this profile review; no architecture claims were extracted from that response."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}},"description":"LC-MS lipid feature filtering and annotation","facets":{"areas":["lipidomics"]},"id":"discovery-model-lipidfinder","kind":"model","links":[],"name":"LipidFinder","source_ids":["src-discovery-lipidfinder"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-massspecgym"],"entity_level":"method","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"matchms","version":null,"profile":{"summary":"matchms is a toolkit for cleaning, processing and comparing tandem mass spectra.","sections":[{"title":"How it works","body":"Spectra and metadata are imported, filtered and validated before a selected pairwise similarity measure is applied. Learned similarity plug-ins are separate methods from the core processing toolkit.","source_ids":["src-discovery-matchms-matchms"],"source_locator":"README.rst: opening description and supported workflows"}],"facts":[{"label":"Method class","value":"Mass-spectral processing and comparison software","source_ids":["src-discovery-matchms-matchms"],"source_locator":"README.rst: opening description and supported workflows"}],"strengths":[{"text":"Makes preprocessing and similarity choices explicit and extensible.","source_ids":["src-discovery-matchms-matchms"],"source_locator":"README.rst: opening description and supported workflows"}],"limitations":[{"text":"Different filtering rules or similarity plug-ins define different pipelines; the toolkit name is not a unique evaluated model.","source_ids":["src-discovery-matchms-matchms"],"source_locator":"README.rst: opening description and supported workflows"}],"diagram":{"title":"Conceptual procedure","steps":["MS/MS files","Metadata and peak cleaning","Similarity measure","Pairwise comparison","Scores and matches"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-matchms-matchms"],"source_locator":"README.rst: opening description and supported workflows"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Mass spectral processing and similarity matching","facets":{"areas":["metabolomics"]},"id":"discovery-model-matchms","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-massspecgym"}],"name":"matchms","source_ids":["src-discovery-matchms-matchms"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-cami"],"entity_level":"method","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"MetaPhlAn","version":null,"profile":{"summary":"MetaPhlAn: marker-based microbial profiling","sections":[{"title":"Available evidence","body":"The discovery record links to biobakery/MetaPhlAn official source. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["src-discovery-biobakery-metaphlan"],"source_locator":"README.md"}],"facts":[],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["src-discovery-biobakery-metaphlan"],"source_locator":"README.md"}],"coverage":"limited","gaps":["The linked README.md has not yielded a reviewed architecture or procedure for the configuration 'MetaPhlAn' in this release.","Unresolved registry fields: checkpoint (unextracted), code licence (unextracted), parameters (unextracted), training cutoff (unextracted), training data (unextracted), version (unextracted), weights licence (unextracted)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}},"description":"Marker-based microbial profiling","facets":{"areas":["microbiome"]},"id":"discovery-model-metaphlan","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-cami"}],"name":"MetaPhlAn","source_ids":["src-discovery-biobakery-metaphlan"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-cafa"],"entity_level":"method","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"MMseqs2","version":null,"profile":{"summary":"MMseqs2 is a toolkit for searching and clustering large protein and nucleotide sequence collections.","sections":[{"title":"How it works","body":"A query collection is searched against a specified sequence or profile database, or clustered using configured similarity criteria. The selected command and database define the operational method.","source_ids":["src-discovery-soedinglab-mmseqs2"],"source_locator":"README.md: introduction and supported workflows"}],"facts":[{"label":"Method class","value":"Sequence search and clustering software","source_ids":["src-discovery-soedinglab-mmseqs2"],"source_locator":"README.md: introduction and supported workflows"}],"strengths":[{"text":"Provides established sequence-similarity procedures that can act as task-specific comparators.","source_ids":["src-discovery-soedinglab-mmseqs2"],"source_locator":"README.md: introduction and supported workflows"}],"limitations":[{"text":"Database contents, coverage thresholds and sensitivity settings must be matched before interpreting comparisons with learned models.","source_ids":["src-discovery-soedinglab-mmseqs2"],"source_locator":"README.md: introduction and supported workflows"}],"diagram":{"title":"Conceptual procedure","steps":["Query sequences","Reference database","Search or clustering","Alignments or clusters"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-soedinglab-mmseqs2"],"source_locator":"README.md: introduction and supported workflows"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Sequence search and clustering","facets":{"areas":["protein-function"]},"id":"discovery-model-mmseqs2","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-cafa"}],"name":"MMseqs2","source_ids":["src-discovery-soedinglab-mmseqs2"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-massspecgym"],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"MSAlign","version":null,"profile":{"summary":"MSAlign retrieves candidate molecules from tandem mass spectra using aligned representations.","sections":[{"title":"How it works","body":"Frozen DreaMS spectrum and ChemBERTa molecule encoders feed lightweight projections. Candidate-based contrastive training aligns their representations; retrieval ranks molecules from the specified candidate set.","source_ids":["src-discovery-msalign"],"source_locator":"arXiv:2605.19752v1 abstract"}],"facts":[{"label":"Evaluated task","value":"Molecule retrieval from MS/MS","source_ids":["src-discovery-msalign"],"source_locator":"arXiv:2605.19752v1 abstract"}],"strengths":[{"text":"Reuses pretrained encoders while training smaller alignment components.","source_ids":["src-discovery-msalign"],"source_locator":"arXiv:2605.19752v1 abstract"}],"limitations":[{"text":"Retrieval difficulty depends on candidate construction and splitting. The paper explicitly examines the tradeoff between leakage and distribution shift.","source_ids":["src-discovery-msalign"],"source_locator":"arXiv:2605.19752v1 abstract"}],"diagram":{"title":"Conceptual procedure","steps":["Spectrum and candidate molecules","Frozen encoders","Projection networks","Shared representation","Candidate ranking"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-msalign"],"source_locator":"arXiv:2605.19752v1 abstract"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Spectrum-to-molecule representation alignment","facets":{"areas":["metabolomics"]},"id":"discovery-model-msalign","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-massspecgym"}],"name":"MSAlign","source_ids":["src-discovery-msalign"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-geneb"],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"Nucleotide Transformer","version":null,"profile":{"summary":"Nucleotide Transformer: genomic representation model family","sections":[{"title":"Available evidence","body":"The discovery record links to instadeepai/nucleotide-transformer official source. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["src-discovery-instadeepai-nucleotide-transformer"],"source_locator":"README.md"}],"facts":[],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["src-discovery-instadeepai-nucleotide-transformer"],"source_locator":"README.md"}],"coverage":"limited","gaps":["The linked README.md has not yielded a reviewed architecture or procedure for the configuration 'Nucleotide Transformer' in this release.","Unresolved registry fields: checkpoint (unextracted), code licence (unextracted), parameters (unextracted), training cutoff (unextracted), training data (unextracted), version (unextracted), weights licence (unextracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}},"description":"Genomic representation model family","facets":{"areas":["genomics"]},"id":"discovery-model-nucleotide-transformer","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-geneb"}],"name":"Nucleotide Transformer","source_ids":["src-discovery-instadeepai-nucleotide-transformer"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-casp"],"entity_level":"method","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"OpenFold","version":null,"profile":{"summary":"OpenFold: trainable protein structure prediction implementation","sections":[{"title":"Available evidence","body":"The discovery record links to aqlaboratory/openfold official source. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["src-discovery-aqlaboratory-openfold"],"source_locator":"README.md"}],"facts":[],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["src-discovery-aqlaboratory-openfold"],"source_locator":"README.md"}],"coverage":"limited","gaps":["The linked README.md has not yielded a reviewed architecture or procedure for the configuration 'OpenFold' in this release.","Unresolved registry fields: checkpoint (unextracted), code licence (unextracted), parameters (unextracted), training cutoff (unextracted), training data (unextracted), version (unextracted), weights licence (unextracted)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}},"description":"Trainable protein structure prediction implementation","facets":{"areas":["protein-structure"]},"id":"discovery-model-openfold","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-casp"}],"name":"OpenFold","source_ids":["src-discovery-aqlaboratory-openfold"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":[],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"Pangolin","version":null,"profile":{"summary":"Pangolin predicts changes in splice-site strength from DNA variants. It accepts variant files or custom sequence inputs.","sections":[{"title":"Architecture","body":"Pangolin uses 16 residual blocks with dilated convolutions and skip connections. Separate outputs estimate splice-site probability and usage across heart, liver, brain and testis. The published model was trained using sequence and splicing measurements from human, rhesus macaque, rat and mouse.","source_ids":["src-discovery-tkzeng-pangolin"],"source_locator":"Original paper linked in README: https://doi.org/10.1186/s13059-022-02664-4, Methods: Deep neural network architecture; Results: Pangolin predicts splice site usage"},{"title":"How it works","body":"Reference genome and transcript annotation define the sequence context. The neural predictor estimates splice-site strength; the command-line tool reports the largest positive and negative changes near each variant. Masking optionally removes particular gains and losses at annotated sites.","source_ids":["src-discovery-tkzeng-pangolin"],"source_locator":"README.md: introduction and Usage (command-line), including supported variants and --mask"}],"facts":[{"label":"Masking","value":"Default mask=True; a mask=False evaluation is a distinct configuration","source_ids":["src-discovery-tkzeng-pangolin"],"source_locator":"README.md: introduction and Usage (command-line), including supported variants and --mask"}],"strengths":[{"text":"Provides changes in splice strength and their positions, with configurable scoring distance and masking.","source_ids":["src-discovery-tkzeng-pangolin"],"source_locator":"README.md: introduction and Usage (command-line), including supported variants and --mask"}],"limitations":[{"text":"Only substitutions and simple indels are supported by the documented interface. Missing gene annotations, reference mismatches and chromosome-edge cases can exclude variants.","source_ids":["src-discovery-tkzeng-pangolin"],"source_locator":"README.md: introduction and Usage (command-line), including supported variants and --mask"}],"diagram":{"title":"Conceptual procedure","steps":["DNA context","Dilated residual convolutions","Tissue-specific outputs","Splice strength","Variant-induced change"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-tkzeng-pangolin"],"source_locator":"Original paper linked in README: https://doi.org/10.1186/s13059-022-02664-4, Figure 1 and Methods: Deep neural network architecture; README Usage"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Splice site strength prediction","facets":{"areas":["genomics"]},"id":"discovery-model-pangolin","kind":"model","links":[],"name":"Pangolin","source_ids":["src-discovery-tkzeng-pangolin"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-proteinbench"],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"ProteinMPNN","version":null,"profile":{"summary":"ProteinMPNN designs amino-acid sequences for a supplied protein backbone, with controls for fixed residues and chains.","sections":[{"title":"How it works","body":"A parsed structure and design constraints are supplied to the sequence-design model. It samples amino-acid sequences conditional on the backbone; sampling temperature changes diversity. Full-backbone and Cα-only weights are separate configurations.","source_ids":["src-discovery-dauparas-proteinmpnn"],"source_locator":"README.md: Full protein backbone models, CA only models and protein_mpnn_run.py arguments"}],"facts":[{"label":"Catalogue weight name","value":"v_48_020","source_ids":["src-discovery-dauparas-proteinmpnn"],"source_locator":"README.md: Full protein backbone models, CA only models and protein_mpnn_run.py arguments"},{"label":"Output","value":"Designed sequences and model scores","source_ids":["src-discovery-dauparas-proteinmpnn"],"source_locator":"README.md: Full protein backbone models, CA only models and protein_mpnn_run.py arguments"}],"strengths":[{"text":"Allows selected chains and positions to be redesigned while retaining specified residues.","source_ids":["src-discovery-dauparas-proteinmpnn"],"source_locator":"README.md: Full protein backbone models, CA only models and protein_mpnn_run.py arguments"}],"limitations":[{"text":"Requires a suitable input structure. Sequence generation does not itself demonstrate folding, activity or experimental success.","source_ids":["src-discovery-dauparas-proteinmpnn"],"source_locator":"README.md: Full protein backbone models, CA only models and protein_mpnn_run.py arguments"}],"diagram":{"title":"Conceptual procedure","steps":["Backbone structure","Chain / residue constraints","ProteinMPNN","Conditional sequence sampling","Designed sequences"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-dauparas-proteinmpnn"],"source_locator":"README.md: Full protein backbone models, CA only models and protein_mpnn_run.py arguments"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Structure-conditioned protein sequence design","facets":{"areas":["protein-structure"]},"id":"discovery-model-proteinmpnn","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-proteinbench"}],"name":"ProteinMPNN","source_ids":["src-discovery-dauparas-proteinmpnn"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-proteinbench"],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"RFdiffusion","version":null,"profile":{"summary":"RFdiffusion generates protein structures with optional conditioning such as motifs or target information.","sections":[{"title":"How it works","body":"A diffusion-based structure-generation workflow samples designs subject to the selected conditioning. Sequence design and experimental testing are subsequent steps rather than guaranteed properties of the generated backbone.","source_ids":["src-discovery-rosettacommons-rfdiffusion"],"source_locator":"README.md: What is RFdiffusion? and supported design challenges"}],"facts":[{"label":"Output","value":"Generated protein structures","source_ids":["src-discovery-rosettacommons-rfdiffusion"],"source_locator":"README.md: What is RFdiffusion? and supported design challenges"}],"strengths":[{"text":"Supports unconditional generation and constrained protein-design problems.","source_ids":["src-discovery-rosettacommons-rfdiffusion"],"source_locator":"README.md: What is RFdiffusion? and supported design challenges"}],"limitations":[{"text":"A generated structure is a candidate design, not evidence of an experimentally functional binder.","source_ids":["src-discovery-rosettacommons-rfdiffusion"],"source_locator":"README.md: What is RFdiffusion? and supported design challenges"}],"diagram":{"title":"Conceptual procedure","steps":["Design constraints","Structure diffusion","Generated backbone","Separate sequence design","Experimental evaluation"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-rosettacommons-rfdiffusion"],"source_locator":"README.md: What is RFdiffusion? and supported design challenges"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Protein structure generation","facets":{"areas":["protein-structure"]},"id":"discovery-model-rfdiffusion","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-proteinbench"}],"name":"RFdiffusion","source_ids":["src-discovery-rosettacommons-rfdiffusion"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-beacon"],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"RNA-FM","version":null,"profile":{"summary":"RNA-FM is a pretrained RNA sequence encoder for structural and functional representation learning.","sections":[{"title":"How it works","body":"A BERT-style transformer encodes RNA tokens into contextual embeddings after self-supervised sequence training. Structural or functional predictions require the corresponding downstream model; RNA-FM alone should not be labelled as a complete 3D folding pipeline.","source_ids":["src-discovery-ml4bio-rna-fm"],"source_locator":"README.md: introduction and Foundation Models and Extended Ecosystem"}],"facts":[{"label":"RNA-FM training","value":"Repository reports more than 23 million non-coding RNA sequences","source_ids":["src-discovery-ml4bio-rna-fm"],"source_locator":"README.md: introduction and Foundation Models and Extended Ecosystem"}],"strengths":[{"text":"Reusable representations do not require experimental labels during pretraining.","source_ids":["src-discovery-ml4bio-rna-fm"],"source_locator":"README.md: introduction and Foundation Models and Extended Ecosystem"}],"limitations":[{"text":"The ncRNA encoder and the coding-sequence mRNA-FM extension have different training modalities and should not share checkpoint identities.","source_ids":["src-discovery-ml4bio-rna-fm"],"source_locator":"README.md: introduction and Foundation Models and Extended Ecosystem"}],"diagram":{"title":"Conceptual procedure","steps":["RNA sequence","RNA tokens","Pretrained transformer","Contextual embeddings","Task-specific predictor"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-ml4bio-rna-fm"],"source_locator":"README.md: introduction and Foundation Models and Extended Ecosystem"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"RNA sequence representation family","facets":{"areas":["rna"]},"id":"discovery-model-rna-fm","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-beacon"}],"name":"RNA-FM","source_ids":["src-discovery-ml4bio-rna-fm"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-perturbench"],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"scGPT","version":null,"profile":{"summary":"scGPT learns representations of genes and cells from single-cell measurements. Its pretrained checkpoints support task-specific adaptation.","sections":[{"title":"How it works","body":"Gene identifiers and expression values are encoded together and processed by a transformer. The resulting representations support cell embeddings or task heads. Vocabulary, preprocessing and the selected checkpoint must accompany any result.","source_ids":["src-discovery-bowang-lab-scgpt"],"source_locator":"README.md: Pretrained scGPT checkpoints and Tutorials; scgpt/model/model.py at cebd6fae655b9c585a4807daa3ac31bb764f06b4, lines 79–188"}],"facts":[{"label":"Whole-human training","value":"Repository reports 33 million normal human cells","source_ids":["src-discovery-bowang-lab-scgpt"],"source_locator":"README.md: Pretrained scGPT checkpoints and Tutorials; scgpt/model/model.py at cebd6fae655b9c585a4807daa3ac31bb764f06b4, lines 79–188"}],"strengths":[{"text":"The repository provides whole-human and specialised checkpoints, plus workflows for annotation, integration and perturbation tasks.","source_ids":["src-discovery-bowang-lab-scgpt"],"source_locator":"README.md: Pretrained scGPT checkpoints and Tutorials; scgpt/model/model.py at cebd6fae655b9c585a4807daa3ac31bb764f06b4, lines 79–188"}],"limitations":[{"text":"Whole-human, organ-specific and continually pretrained checkpoints are different configurations. A pretraining claim does not establish transfer performance in a new cell population.","source_ids":["src-discovery-bowang-lab-scgpt"],"source_locator":"README.md: Pretrained scGPT checkpoints and Tutorials; scgpt/model/model.py at cebd6fae655b9c585a4807daa3ac31bb764f06b4, lines 79–188"}],"diagram":{"title":"Conceptual procedure","steps":["Gene IDs and expression","Gene / value encoders","Transformer","Cell and gene representations","Adapted task output"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-bowang-lab-scgpt"],"source_locator":"README.md: Pretrained scGPT checkpoints and Tutorials; scgpt/model/model.py at cebd6fae655b9c585a4807daa3ac31bb764f06b4, lines 79–188"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Single-cell multi-omics model","facets":{"areas":["single-cell"]},"id":"discovery-model-scgpt","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-perturbench"}],"name":"scGPT","source_ids":["src-discovery-bowang-lab-scgpt"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-scib"],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"scVI","version":null,"profile":{"summary":"scVI: probabilistic single-cell expression model","sections":[{"title":"Available evidence","body":"The discovery record links to scverse/scvi-tools official source. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["src-discovery-scverse-scvi-tools"],"source_locator":"README.md"}],"facts":[],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["src-discovery-scverse-scvi-tools"],"source_locator":"README.md"}],"coverage":"limited","gaps":["The linked README.md has not yielded a reviewed architecture or procedure for the configuration 'scVI' in this release.","Unresolved registry fields: checkpoint (unextracted), code licence (unextracted), parameters (unextracted), training cutoff (unextracted), training data (unextracted), version (unextracted), weights licence (unextracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}},"description":"Probabilistic single-cell expression model","facets":{"areas":["single-cell"]},"id":"discovery-model-scvi","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-scib"}],"name":"scVI","source_ids":["src-discovery-scverse-scvi-tools"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":[],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"SegmentNT","version":null,"profile":{"summary":"SegmentNT: genomic sequence segmentation model","sections":[{"title":"Available evidence","body":"The discovery record links to instadeepai/nucleotide-transformer official source. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["src-discovery-instadeepai-nucleotide-transformer"],"source_locator":"README.md"}],"facts":[],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["src-discovery-instadeepai-nucleotide-transformer"],"source_locator":"README.md"}],"coverage":"limited","gaps":["The linked README.md has not yielded a reviewed architecture or procedure for the configuration 'SegmentNT' in this release.","Unresolved registry fields: checkpoint (unextracted), code licence (unextracted), parameters (unextracted), training cutoff (unextracted), training data (unextracted), version (unextracted), weights licence (unextracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}},"description":"Genomic sequence segmentation model","facets":{"areas":["genomics"]},"id":"discovery-model-segmentnt","kind":"model","links":[],"name":"SegmentNT","source_ids":["src-discovery-instadeepai-nucleotide-transformer"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":[],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"SpliceAI","version":null,"profile":{"summary":"SpliceAI predicts how sequence variants alter splice-site usage. Its variant annotation tool combines reference sequence with gene annotation.","sections":[{"title":"Architecture","body":"SpliceAI uses a residual convolutional network with dilated filters to integrate sequence context. It predicts donor, acceptor and non-splice-site probabilities along the sequence; comparing alleles converts those predictions into variant scores. The output is not tissue-specific.","source_ids":["src-discovery-illumina-spliceai","src-discovery-tkzeng-pangolin"],"source_locator":"SpliceAI README and linked Jaganathan et al. paper; Pangolin primary paper https://doi.org/10.1186/s13059-022-02664-4, Methods: Deep neural network architecture, direct comparison with SpliceAI"},{"title":"How it works","body":"The tool evaluates reference and alternative alleles and reports predicted acceptor/donor gains and losses with their relative positions. Annotation, search distance and masking determine the reported variant scores.","source_ids":["src-discovery-illumina-spliceai"],"source_locator":"README.md: package description, License, Usage and Frequently Asked Questions"}],"facts":[{"label":"Inputs","value":"VCF, reference FASTA and gene annotation","source_ids":["src-discovery-illumina-spliceai"],"source_locator":"README.md: package description, License, Usage and Frequently Asked Questions"},{"label":"Outputs","value":"Acceptor/donor gain and loss delta scores","source_ids":["src-discovery-illumina-spliceai"],"source_locator":"README.md: package description, License, Usage and Frequently Asked Questions"}],"strengths":[{"text":"Produces splice-specific scores and predicted event positions without fitting a classifier to the user’s assay labels.","source_ids":["src-discovery-illumina-spliceai"],"source_locator":"README.md: package description, License, Usage and Frequently Asked Questions"}],"limitations":[{"text":"Code, trained models and precomputed annotations have distinct use terms. Annotation and sequence checks can leave variants unscored.","source_ids":["src-discovery-illumina-spliceai"],"source_locator":"README.md: package description, License, Usage and Frequently Asked Questions"}],"diagram":{"title":"Conceptual procedure","steps":["Reference / alternate DNA","Dilated residual convolutions","Acceptor / donor probabilities","Allelic difference","Variant delta scores"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-illumina-spliceai","src-discovery-tkzeng-pangolin"],"source_locator":"SpliceAI README and linked Jaganathan et al. paper; Pangolin primary paper https://doi.org/10.1186/s13059-022-02664-4, Methods: Deep neural network architecture, direct comparison with SpliceAI"},"coverage":"reviewed","gaps":["An immutable hash for the exact 1.3.1 weight files has not been attached to this family record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Splicing effect prediction","facets":{"areas":["genomics"]},"id":"discovery-model-spliceai","kind":"model","links":[],"name":"SpliceAI","source_ids":["src-discovery-illumina-spliceai"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-virtual-cell-challenge-2026"],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"STATE","version":null,"profile":{"summary":"STATE: cell state and perturbation modelling","sections":[{"title":"Available evidence","body":"The discovery record links to ArcInstitute/state official source. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["src-discovery-arcinstitute-state"],"source_locator":"README.md"}],"facts":[],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["src-discovery-arcinstitute-state"],"source_locator":"README.md"}],"coverage":"limited","gaps":["The linked README.md has not yielded a reviewed architecture or procedure for the configuration 'STATE' in this release.","Unresolved registry fields: checkpoint (unextracted), code licence (unextracted), parameters (unextracted), training cutoff (unextracted), training data (unextracted), version (unextracted), weights licence (unextracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}},"description":"Cell state and perturbation modelling","facets":{"areas":["single-cell"]},"id":"discovery-model-state","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-virtual-cell-challenge-2026"}],"name":"STATE","source_ids":["src-discovery-arcinstitute-state"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-glycanml"],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"SweetNet","version":null,"profile":{"summary":"SweetNet: glycan graph learning model","sections":[{"title":"Available evidence","body":"The discovery record links to BojarLab/glycowork official source. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["src-discovery-bojarlab-glycowork"],"source_locator":"README.md"}],"facts":[],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["src-discovery-bojarlab-glycowork"],"source_locator":"README.md"}],"coverage":"limited","gaps":["The linked README.md has not yielded a reviewed architecture or procedure for the configuration 'SweetNet' in this release.","Unresolved registry fields: checkpoint (unextracted), code licence (unextracted), parameters (unextracted), training cutoff (unextracted), training data (unextracted), version (unextracted), weights licence (unextracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}},"description":"Glycan graph learning model","facets":{"areas":["glycomics"]},"id":"discovery-model-sweetnet","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-glycanml"}],"name":"SweetNet","source_ids":["src-discovery-bojarlab-glycowork"],"status":"discovered"} {"attributes":{"entity_level":"method","missing_metadata":{"checkpoint":"unreported","version":"unreported"},"reported_name":"Bepler","version":null,"profile":{"summary":"TAPE Bepler is the method recorded for TAPE Fluorescence Bepler leaderboard evaluation. This page preserves the configuration reported by songlab-cal/tape official source.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: TAPE Fluorescence TAPE Stability","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README.md > Leaderboard > Fluorescence; row Bepler; column Spearman's rho; README.md > Leaderboard > Stability; row Bepler; column Spearman's rho"}],"facts":[{"label":"Recorded dataset","value":"TAPE Fluorescence source dataset","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README.md > Leaderboard > Fluorescence; row Bepler; column Spearman's rho; README.md > Leaderboard > Stability; row Bepler; column Spearman's rho"},{"label":"Recorded dataset","value":"TAPE Stability source dataset","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README.md > Leaderboard > Fluorescence; row Bepler; column Spearman's rho; README.md > Leaderboard > Stability; row Bepler; column Spearman's rho"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README.md > Leaderboard > Fluorescence; row Bepler; column Spearman's rho; README.md > Leaderboard > Stability; row Bepler; column Spearman's rho"}],"coverage":"limited","gaps":["The retained evidence location (README.md > Leaderboard > Fluorescence; row Bepler; column Spearman's rho; README.md > Leaderboard > Stability; row Bepler; column Spearman's rho) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: checkpoint (unreported), version (unreported)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}},"description":"Method identity as printed in the historical TAPE leaderboard.","facets":{"areas":["protein-function"]},"id":"discovery-model-tape-bepler","kind":"model","links":[],"name":"TAPE Bepler","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"entity_level":"method","missing_metadata":{"checkpoint":"unreported","version":"unreported"},"reported_name":"LSTM","version":null,"profile":{"summary":"TAPE LSTM is the method recorded for TAPE Fluorescence LSTM leaderboard evaluation. This page preserves the configuration reported by songlab-cal/tape official source.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: TAPE Fluorescence TAPE Stability","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README.md > Leaderboard > Fluorescence; row LSTM; column Spearman's rho; README.md > Leaderboard > Stability; row LSTM; column Spearman's rho"}],"facts":[{"label":"Recorded dataset","value":"TAPE Fluorescence source dataset","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README.md > Leaderboard > Fluorescence; row LSTM; column Spearman's rho; README.md > Leaderboard > Stability; row LSTM; column Spearman's rho"},{"label":"Recorded dataset","value":"TAPE Stability source dataset","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README.md > Leaderboard > Fluorescence; row LSTM; column Spearman's rho; README.md > Leaderboard > Stability; row LSTM; column Spearman's rho"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README.md > Leaderboard > Fluorescence; row LSTM; column Spearman's rho; README.md > Leaderboard > Stability; row LSTM; column Spearman's rho"}],"coverage":"limited","gaps":["The retained evidence location (README.md > Leaderboard > Fluorescence; row LSTM; column Spearman's rho; README.md > Leaderboard > Stability; row LSTM; column Spearman's rho) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: checkpoint (unreported), version (unreported)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}},"description":"Method identity as printed in the historical TAPE leaderboard.","facets":{"areas":["protein-function"]},"id":"discovery-model-tape-lstm","kind":"model","links":[],"name":"TAPE LSTM","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"entity_level":"method","missing_metadata":{"checkpoint":"unreported","version":"unreported"},"reported_name":"One Hot","version":null,"profile":{"summary":"TAPE One Hot is the method recorded for TAPE Fluorescence One Hot leaderboard evaluation. This page preserves the configuration reported by songlab-cal/tape official source.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: TAPE Fluorescence TAPE Stability","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README.md > Leaderboard > Fluorescence; row One Hot; column Spearman's rho; README.md > Leaderboard > Stability; row One Hot; column Spearman's rho"}],"facts":[{"label":"Recorded dataset","value":"TAPE Fluorescence source dataset","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README.md > Leaderboard > Fluorescence; row One Hot; column Spearman's rho; README.md > Leaderboard > Stability; row One Hot; column Spearman's rho"},{"label":"Recorded dataset","value":"TAPE Stability source dataset","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README.md > Leaderboard > Fluorescence; row One Hot; column Spearman's rho; README.md > Leaderboard > Stability; row One Hot; column Spearman's rho"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README.md > Leaderboard > Fluorescence; row One Hot; column Spearman's rho; README.md > Leaderboard > Stability; row One Hot; column Spearman's rho"}],"coverage":"limited","gaps":["The retained evidence location (README.md > Leaderboard > Fluorescence; row One Hot; column Spearman's rho; README.md > Leaderboard > Stability; row One Hot; column Spearman's rho) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: checkpoint (unreported), version (unreported)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}},"description":"Method identity as printed in the historical TAPE leaderboard.","facets":{"areas":["protein-function"]},"id":"discovery-model-tape-one-hot","kind":"model","links":[],"name":"TAPE One Hot","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"entity_level":"method","missing_metadata":{"checkpoint":"unreported","version":"unreported"},"reported_name":"ResNet","version":null,"profile":{"summary":"TAPE ResNet is the method recorded for TAPE Fluorescence ResNet leaderboard evaluation. This page preserves the configuration reported by songlab-cal/tape official source.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: TAPE Fluorescence TAPE Stability","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README.md > Leaderboard > Fluorescence; row ResNet; column Spearman's rho; README.md > Leaderboard > Stability; row ResNet; column Spearman's rho"}],"facts":[{"label":"Recorded dataset","value":"TAPE Fluorescence source dataset","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README.md > Leaderboard > Fluorescence; row ResNet; column Spearman's rho; README.md > Leaderboard > Stability; row ResNet; column Spearman's rho"},{"label":"Recorded dataset","value":"TAPE Stability source dataset","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README.md > Leaderboard > Fluorescence; row ResNet; column Spearman's rho; README.md > Leaderboard > Stability; row ResNet; column Spearman's rho"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README.md > Leaderboard > Fluorescence; row ResNet; column Spearman's rho; README.md > Leaderboard > Stability; row ResNet; column Spearman's rho"}],"coverage":"limited","gaps":["The retained evidence location (README.md > Leaderboard > Fluorescence; row ResNet; column Spearman's rho; README.md > Leaderboard > Stability; row ResNet; column Spearman's rho) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: checkpoint (unreported), version (unreported)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}},"description":"Method identity as printed in the historical TAPE leaderboard.","facets":{"areas":["protein-function"]},"id":"discovery-model-tape-resnet","kind":"model","links":[],"name":"TAPE ResNet","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"entity_level":"method","missing_metadata":{"checkpoint":"unreported","version":"unreported"},"reported_name":"Transformer","version":null,"profile":{"summary":"TAPE Transformer is the method recorded for TAPE Fluorescence Transformer leaderboard evaluation. This page preserves the configuration reported by songlab-cal/tape official source.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: TAPE Fluorescence TAPE Stability","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README.md > Leaderboard > Fluorescence; row Transformer; column Spearman's rho; README.md > Leaderboard > Stability; row Transformer; column Spearman's rho"}],"facts":[{"label":"Recorded dataset","value":"TAPE Fluorescence source dataset","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README.md > Leaderboard > Fluorescence; row Transformer; column Spearman's rho; README.md > Leaderboard > Stability; row Transformer; column Spearman's rho"},{"label":"Recorded dataset","value":"TAPE Stability source dataset","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README.md > Leaderboard > Fluorescence; row Transformer; column Spearman's rho; README.md > Leaderboard > Stability; row Transformer; column Spearman's rho"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README.md > Leaderboard > Fluorescence; row Transformer; column Spearman's rho; README.md > Leaderboard > Stability; row Transformer; column Spearman's rho"}],"coverage":"limited","gaps":["The retained evidence location (README.md > Leaderboard > Fluorescence; row Transformer; column Spearman's rho; README.md > Leaderboard > Stability; row Transformer; column Spearman's rho) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: checkpoint (unreported), version (unreported)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}},"description":"Method identity as printed in the historical TAPE leaderboard.","facets":{"areas":["protein-function"]},"id":"discovery-model-tape-transformer","kind":"model","links":[],"name":"TAPE Transformer","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"entity_level":"method","missing_metadata":{"checkpoint":"unreported","version":"unreported"},"reported_name":"Unirep","version":null,"profile":{"summary":"TAPE Unirep is the method recorded for TAPE Fluorescence Unirep leaderboard evaluation. This page preserves the configuration reported by songlab-cal/tape official source.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: TAPE Fluorescence TAPE Stability","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README.md > Leaderboard > Fluorescence; row Unirep; column Spearman's rho; README.md > Leaderboard > Stability; row Unirep; column Spearman's rho"}],"facts":[{"label":"Recorded dataset","value":"TAPE Fluorescence source dataset","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README.md > Leaderboard > Fluorescence; row Unirep; column Spearman's rho; README.md > Leaderboard > Stability; row Unirep; column Spearman's rho"},{"label":"Recorded dataset","value":"TAPE Stability source dataset","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README.md > Leaderboard > Fluorescence; row Unirep; column Spearman's rho; README.md > Leaderboard > Stability; row Unirep; column Spearman's rho"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README.md > Leaderboard > Fluorescence; row Unirep; column Spearman's rho; README.md > Leaderboard > Stability; row Unirep; column Spearman's rho"}],"coverage":"limited","gaps":["The retained evidence location (README.md > Leaderboard > Fluorescence; row Unirep; column Spearman's rho; README.md > Leaderboard > Stability; row Unirep; column Spearman's rho) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: checkpoint (unreported), version (unreported)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}},"description":"Method identity as printed in the historical TAPE leaderboard.","facets":{"areas":["protein-function"]},"id":"discovery-model-tape-unirep","kind":"model","links":[],"name":"TAPE Unirep","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-beacon"],"entity_level":"method","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"ViennaRNA RNAfold","version":null,"profile":{"summary":"RNAfold predicts RNA secondary structure using thermodynamic calculations in the ViennaRNA package.","sections":[{"title":"How it works","body":"An RNA sequence is processed using the selected energy model. The program calculates a minimum-free-energy secondary structure and can compute a partition function.","source_ids":["src-discovery-viennarna-viennarna"],"source_locator":"README.md: Program descriptions, RNAfold row"}],"facts":[{"label":"Method class","value":"Thermodynamic RNA folding","source_ids":["src-discovery-viennarna-viennarna"],"source_locator":"README.md: Program descriptions, RNAfold row"}],"strengths":[{"text":"Provides a mechanistic reference for comparison with learned RNA-structure predictors.","source_ids":["src-discovery-viennarna-viennarna"],"source_locator":"README.md: Program descriptions, RNAfold row"}],"limitations":[{"text":"The energy parameterisation and folding options belong to the evaluated configuration; secondary structure is not a complete 3D structure.","source_ids":["src-discovery-viennarna-viennarna"],"source_locator":"README.md: Program descriptions, RNAfold row"}],"diagram":{"title":"Conceptual procedure","steps":["RNA sequence","Thermodynamic model","Energy / partition calculation","Secondary-structure prediction"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-viennarna-viennarna"],"source_locator":"README.md: Program descriptions, RNAfold row"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Thermodynamic RNA secondary structure prediction","facets":{"areas":["rna"]},"id":"discovery-model-viennarna-rnafold","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-beacon"}],"name":"ViennaRNA RNAfold","source_ids":["src-discovery-viennarna-viennarna"],"status":"discovered"} {"attributes":{"metric":"Spearman's rho","metric_direction":"higher","missing_metadata":{"denominator":"unextracted","seeds":"unreported","uncertainty":"unreported"},"numeric_value":"0.33","printed_value":"0.33","review":{"method":"automated table parse plus separate source read and manually enumerated row/value cross-check","notes":"All six rows checked for this task. Source checked, not experiment reproduced. Historical README does not resolve exact checkpoint or data release.","reviewed_at":"2026-09-16T10:35:25.806835+00:00","reviewer":"Codex research agent; no human review claimed"},"source_locator":"README.md > Leaderboard > Fluorescence; row Bepler; column Spearman's rho","uncertainty":null,"unit":"correlation"},"description":"Reported score transcribed from the complete historical task table.","facets":{"areas":["protein-function"]},"id":"discovery-result-tape-fluorescence-bepler-spearman-rho","kind":"result","links":[{"relation":"evaluation","target_id":"discovery-evaluation-tape-fluorescence-bepler-leaderboard-evaluation"}],"name":"TAPE Fluorescence Bepler Spearman rho","source_ids":["src-discovery-songlab-cal-tape"],"status":"source_checked"} {"attributes":{"metric":"Spearman's rho","metric_direction":"higher","missing_metadata":{"denominator":"unextracted","seeds":"unreported","uncertainty":"unreported"},"numeric_value":"0.67","printed_value":"0.67","review":{"method":"automated table parse plus separate source read and manually enumerated row/value cross-check","notes":"All six rows checked for this task. Source checked, not experiment reproduced. Historical README does not resolve exact checkpoint or data release.","reviewed_at":"2026-09-16T10:35:25.806835+00:00","reviewer":"Codex research agent; no human review claimed"},"source_locator":"README.md > Leaderboard > Fluorescence; row LSTM; column Spearman's rho","uncertainty":null,"unit":"correlation"},"description":"Reported score transcribed from the complete historical task table.","facets":{"areas":["protein-function"]},"id":"discovery-result-tape-fluorescence-lstm-spearman-rho","kind":"result","links":[{"relation":"evaluation","target_id":"discovery-evaluation-tape-fluorescence-lstm-leaderboard-evaluation"}],"name":"TAPE Fluorescence LSTM Spearman rho","source_ids":["src-discovery-songlab-cal-tape"],"status":"source_checked"} {"attributes":{"metric":"Spearman's rho","metric_direction":"higher","missing_metadata":{"denominator":"unextracted","seeds":"unreported","uncertainty":"unreported"},"numeric_value":"0.14","printed_value":"0.14","review":{"method":"automated table parse plus separate source read and manually enumerated row/value cross-check","notes":"All six rows checked for this task. Source checked, not experiment reproduced. Historical README does not resolve exact checkpoint or data release.","reviewed_at":"2026-09-16T10:35:25.806835+00:00","reviewer":"Codex research agent; no human review claimed"},"source_locator":"README.md > Leaderboard > Fluorescence; row One Hot; column Spearman's rho","uncertainty":null,"unit":"correlation"},"description":"Reported score transcribed from the complete historical task table.","facets":{"areas":["protein-function"]},"id":"discovery-result-tape-fluorescence-one-hot-spearman-rho","kind":"result","links":[{"relation":"evaluation","target_id":"discovery-evaluation-tape-fluorescence-one-hot-leaderboard-evaluation"}],"name":"TAPE Fluorescence One Hot Spearman rho","source_ids":["src-discovery-songlab-cal-tape"],"status":"source_checked"} {"attributes":{"metric":"Spearman's rho","metric_direction":"higher","missing_metadata":{"denominator":"unextracted","seeds":"unreported","uncertainty":"unreported"},"numeric_value":"0.21","printed_value":"0.21","review":{"method":"automated table parse plus separate source read and manually enumerated row/value cross-check","notes":"All six rows checked for this task. Source checked, not experiment reproduced. Historical README does not resolve exact checkpoint or data release.","reviewed_at":"2026-09-16T10:35:25.806835+00:00","reviewer":"Codex research agent; no human review claimed"},"source_locator":"README.md > Leaderboard > Fluorescence; row ResNet; column Spearman's rho","uncertainty":null,"unit":"correlation"},"description":"Reported score transcribed from the complete historical task table.","facets":{"areas":["protein-function"]},"id":"discovery-result-tape-fluorescence-resnet-spearman-rho","kind":"result","links":[{"relation":"evaluation","target_id":"discovery-evaluation-tape-fluorescence-resnet-leaderboard-evaluation"}],"name":"TAPE Fluorescence ResNet Spearman rho","source_ids":["src-discovery-songlab-cal-tape"],"status":"source_checked"} {"attributes":{"metric":"Spearman's rho","metric_direction":"higher","missing_metadata":{"denominator":"unextracted","seeds":"unreported","uncertainty":"unreported"},"numeric_value":"0.68","printed_value":"0.68","review":{"method":"automated table parse plus separate source read and manually enumerated row/value cross-check","notes":"All six rows checked for this task. Source checked, not experiment reproduced. Historical README does not resolve exact checkpoint or data release.","reviewed_at":"2026-09-16T10:35:25.806835+00:00","reviewer":"Codex research agent; no human review claimed"},"source_locator":"README.md > Leaderboard > Fluorescence; row Transformer; column Spearman's rho","uncertainty":null,"unit":"correlation"},"description":"Reported score transcribed from the complete historical task table.","facets":{"areas":["protein-function"]},"id":"discovery-result-tape-fluorescence-transformer-spearman-rho","kind":"result","links":[{"relation":"evaluation","target_id":"discovery-evaluation-tape-fluorescence-transformer-leaderboard-evaluation"}],"name":"TAPE Fluorescence Transformer Spearman rho","source_ids":["src-discovery-songlab-cal-tape"],"status":"source_checked"} {"attributes":{"metric":"Spearman's rho","metric_direction":"higher","missing_metadata":{"denominator":"unextracted","seeds":"unreported","uncertainty":"unreported"},"numeric_value":"0.67","printed_value":"0.67","review":{"method":"automated table parse plus separate source read and manually enumerated row/value cross-check","notes":"All six rows checked for this task. Source checked, not experiment reproduced. Historical README does not resolve exact checkpoint or data release.","reviewed_at":"2026-09-16T10:35:25.806835+00:00","reviewer":"Codex research agent; no human review claimed"},"source_locator":"README.md > Leaderboard > Fluorescence; row Unirep; column Spearman's rho","uncertainty":null,"unit":"correlation"},"description":"Reported score transcribed from the complete historical task table.","facets":{"areas":["protein-function"]},"id":"discovery-result-tape-fluorescence-unirep-spearman-rho","kind":"result","links":[{"relation":"evaluation","target_id":"discovery-evaluation-tape-fluorescence-unirep-leaderboard-evaluation"}],"name":"TAPE Fluorescence Unirep Spearman rho","source_ids":["src-discovery-songlab-cal-tape"],"status":"source_checked"} {"attributes":{"metric":"Spearman's rho","metric_direction":"higher","missing_metadata":{"denominator":"unextracted","seeds":"unreported","uncertainty":"unreported"},"numeric_value":"0.64","printed_value":"0.64","review":{"method":"automated table parse plus separate source read and manually enumerated row/value cross-check","notes":"All six rows checked for this task. Source checked, not experiment reproduced. Historical README does not resolve exact checkpoint or data release.","reviewed_at":"2026-09-16T10:35:25.806835+00:00","reviewer":"Codex research agent; no human review claimed"},"source_locator":"README.md > Leaderboard > Stability; row Bepler; column Spearman's rho","uncertainty":null,"unit":"correlation"},"description":"Reported score transcribed from the complete historical task table.","facets":{"areas":["protein-function"]},"id":"discovery-result-tape-stability-bepler-spearman-rho","kind":"result","links":[{"relation":"evaluation","target_id":"discovery-evaluation-tape-stability-bepler-leaderboard-evaluation"}],"name":"TAPE Stability Bepler Spearman rho","source_ids":["src-discovery-songlab-cal-tape"],"status":"source_checked"} {"attributes":{"metric":"Spearman's rho","metric_direction":"higher","missing_metadata":{"denominator":"unextracted","seeds":"unreported","uncertainty":"unreported"},"numeric_value":"0.69","printed_value":"0.69","review":{"method":"automated table parse plus separate source read and manually enumerated row/value cross-check","notes":"All six rows checked for this task. Source checked, not experiment reproduced. Historical README does not resolve exact checkpoint or data release.","reviewed_at":"2026-09-16T10:35:25.806835+00:00","reviewer":"Codex research agent; no human review claimed"},"source_locator":"README.md > Leaderboard > Stability; row LSTM; column Spearman's rho","uncertainty":null,"unit":"correlation"},"description":"Reported score transcribed from the complete historical task table.","facets":{"areas":["protein-function"]},"id":"discovery-result-tape-stability-lstm-spearman-rho","kind":"result","links":[{"relation":"evaluation","target_id":"discovery-evaluation-tape-stability-lstm-leaderboard-evaluation"}],"name":"TAPE Stability LSTM Spearman rho","source_ids":["src-discovery-songlab-cal-tape"],"status":"source_checked"} {"attributes":{"metric":"Spearman's rho","metric_direction":"higher","missing_metadata":{"denominator":"unextracted","seeds":"unreported","uncertainty":"unreported"},"numeric_value":"0.19","printed_value":"0.19","review":{"method":"automated table parse plus separate source read and manually enumerated row/value cross-check","notes":"All six rows checked for this task. Source checked, not experiment reproduced. Historical README does not resolve exact checkpoint or data release.","reviewed_at":"2026-09-16T10:35:25.806835+00:00","reviewer":"Codex research agent; no human review claimed"},"source_locator":"README.md > Leaderboard > Stability; row One Hot; column Spearman's rho","uncertainty":null,"unit":"correlation"},"description":"Reported score transcribed from the complete historical task table.","facets":{"areas":["protein-function"]},"id":"discovery-result-tape-stability-one-hot-spearman-rho","kind":"result","links":[{"relation":"evaluation","target_id":"discovery-evaluation-tape-stability-one-hot-leaderboard-evaluation"}],"name":"TAPE Stability One Hot Spearman rho","source_ids":["src-discovery-songlab-cal-tape"],"status":"source_checked"} {"attributes":{"metric":"Spearman's rho","metric_direction":"higher","missing_metadata":{"denominator":"unextracted","seeds":"unreported","uncertainty":"unreported"},"numeric_value":"0.73","printed_value":"0.73","review":{"method":"automated table parse plus separate source read and manually enumerated row/value cross-check","notes":"All six rows checked for this task. Source checked, not experiment reproduced. Historical README does not resolve exact checkpoint or data release.","reviewed_at":"2026-09-16T10:35:25.806835+00:00","reviewer":"Codex research agent; no human review claimed"},"source_locator":"README.md > Leaderboard > Stability; row ResNet; column Spearman's rho","uncertainty":null,"unit":"correlation"},"description":"Reported score transcribed from the complete historical task table.","facets":{"areas":["protein-function"]},"id":"discovery-result-tape-stability-resnet-spearman-rho","kind":"result","links":[{"relation":"evaluation","target_id":"discovery-evaluation-tape-stability-resnet-leaderboard-evaluation"}],"name":"TAPE Stability ResNet Spearman rho","source_ids":["src-discovery-songlab-cal-tape"],"status":"source_checked"} {"attributes":{"metric":"Spearman's rho","metric_direction":"higher","missing_metadata":{"denominator":"unextracted","seeds":"unreported","uncertainty":"unreported"},"numeric_value":"0.73","printed_value":"0.73","review":{"method":"automated table parse plus separate source read and manually enumerated row/value cross-check","notes":"All six rows checked for this task. Source checked, not experiment reproduced. Historical README does not resolve exact checkpoint or data release.","reviewed_at":"2026-09-16T10:35:25.806835+00:00","reviewer":"Codex research agent; no human review claimed"},"source_locator":"README.md > Leaderboard > Stability; row Transformer; column Spearman's rho","uncertainty":null,"unit":"correlation"},"description":"Reported score transcribed from the complete historical task table.","facets":{"areas":["protein-function"]},"id":"discovery-result-tape-stability-transformer-spearman-rho","kind":"result","links":[{"relation":"evaluation","target_id":"discovery-evaluation-tape-stability-transformer-leaderboard-evaluation"}],"name":"TAPE Stability Transformer Spearman rho","source_ids":["src-discovery-songlab-cal-tape"],"status":"source_checked"} {"attributes":{"metric":"Spearman's rho","metric_direction":"higher","missing_metadata":{"denominator":"unextracted","seeds":"unreported","uncertainty":"unreported"},"numeric_value":"0.73","printed_value":"0.73","review":{"method":"automated table parse plus separate source read and manually enumerated row/value cross-check","notes":"All six rows checked for this task. Source checked, not experiment reproduced. Historical README does not resolve exact checkpoint or data release.","reviewed_at":"2026-09-16T10:35:25.806835+00:00","reviewer":"Codex research agent; no human review claimed"},"source_locator":"README.md > Leaderboard > Stability; row Unirep; column Spearman's rho","uncertainty":null,"unit":"correlation"},"description":"Reported score transcribed from the complete historical task table.","facets":{"areas":["protein-function"]},"id":"discovery-result-tape-stability-unirep-spearman-rho","kind":"result","links":[{"relation":"evaluation","target_id":"discovery-evaluation-tape-stability-unirep-leaderboard-evaluation"}],"name":"TAPE Stability Unirep Spearman rho","source_ids":["src-discovery-songlab-cal-tape"],"status":"source_checked"} {"id":"dna-foundation-models-2025","kind":"source","name":"Benchmarking DNA foundation models for genomic and genetic tasks","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["dna-genomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12663285/","version":"PMC12663285.1","retrieved_at":"2026-09-15T23:25:00Z","doi":"10.1038/s41467-025-65823-8","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"5d8ca9bcf88cc1b38ad667906a2e4699b1aefa6d31c6f49259784930353f3202","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12663285/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:35.327Z","legacy_paper":{"id":"dna-foundation-models-2025","title":"Benchmarking DNA foundation models for genomic and genetic tasks","year":2025,"publication_status":"peer_reviewed","version":"PMC12663285.1","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12663285/","primary_domain":"dna-genomes","retrieved_utc":"2026-09-15T23:25:00Z","notes":"Primary full text via Europe PMC XML; venue: Nature Communications; PMC ID: PMC12663285.","doi":"10.1038/s41467-025-65823-8"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"dnabert2-enhancer-2025","kind":"source","name":"Utilizing a deep learning model based on BERT for identifying enhancers and their strength","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["dna-genomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11981215/","version":"journal full text in PMC","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.1371/journal.pone.0320085","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"d052b80efe7bfc1380994ad28503a5575f04ef940f74d5c9c137cb4ba6827863","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11981215/fullTextXML","artifact_retrieved_at":"2026-09-16T10:38:57.558204+00:00","legacy_paper":{"id":"dnabert2-enhancer-2025","title":"Utilizing a deep learning model based on BERT for identifying enhancers and their strength","year":2025,"publication_status":"peer_reviewed","version":"journal full text in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11981215/","primary_domain":"dna-genomes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.1371/journal.pone.0320085","notes":"Numeric result checked against Table 4 in primary full-text XML; journal/source: PLOS One."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"dnalongbench-2025","kind":"source","name":"DNALongBench: A Benchmark Suite for Long-Range DNA Prediction Tasks","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["dna-genomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11741265/","version":"PMC11741265.1","retrieved_at":"2026-09-15T23:25:00Z","doi":"10.1101/2025.01.06.631595","publication_status":"preprint","year":2025,"artifact_sha256":"fa440a17cecf16a5d872d50a30910f7591b5f6f78e10a944c6bda5ea8d7e32dd","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11741265/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:16.492545+00:00","legacy_paper":{"id":"dnalongbench-2025","title":"DNALongBench: A Benchmark Suite for Long-Range DNA Prediction Tasks","year":2025,"publication_status":"preprint","version":"PMC11741265.1","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11741265/","primary_domain":"dna-genomes","retrieved_utc":"2026-09-15T23:25:00Z","notes":"Primary full text via Europe PMC XML; venue: bioRxiv; PMC ID: PMC11741265.","doi":"10.1101/2025.01.06.631595"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"eden-genomic-classification-2026","kind":"source","name":"EDEN: multiscale expected density of nucleotide encoding for enhanced DNA sequence classification with hybrid deep learning","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["dna-genomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12879454/","version":"journal full text in PMC","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.1186/s12859-026-06367-6","publication_status":"peer_reviewed","year":2026,"artifact_sha256":"38a6e26b3caffe8e021a2b0b672218e783aca9ee42046765e323946813015e65","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12879454/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:37.531Z","legacy_paper":{"id":"eden-genomic-classification-2026","title":"EDEN: multiscale expected density of nucleotide encoding for enhanced DNA sequence classification with hybrid deep learning","year":2026,"publication_status":"peer_reviewed","version":"journal full text in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12879454/","primary_domain":"dna-genomes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.1186/s12859-026-06367-6","notes":"DNABERT-2 comparator 70.52 is printed in Table 5. The article does not clearly document whether this comparator was independently rerun or consolidated from prior GUE results, so evaluation origin is conservatively marked paper_compilation."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"enbed-2024","kind":"source","name":"Understanding the natural language of DNA using encoder–decoder foundation models with byte-level precision","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["dna-genomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11341122/","version":"version of record","retrieved_at":"2026-09-15T23:25:00Z","doi":"10.1093/bioadv/vbae117","publication_status":"peer_reviewed","year":2024,"artifact_sha256":"e95d4be70d32e61af5a92eda8ea66f25a2cc629e3f83e7b5241b13cde8bdb83b","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11341122/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:35.378Z","legacy_paper":{"id":"enbed-2024","title":"Understanding the natural language of DNA using encoder–decoder foundation models with byte-level precision","year":2024,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11341122/","primary_domain":"dna-genomes","retrieved_utc":"2026-09-15T23:25:00Z","notes":"Primary full text via Europe PMC XML; venue: Bioinformatics Advances; PMC ID: PMC11341122.","doi":"10.1093/bioadv/vbae117"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"enhancer-position-encoding-2024","kind":"source","name":"A deep learning model for DNA enhancer prediction based on nucleotide position aware feature encoding","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["dna-genomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11167433/","version":"version of record","retrieved_at":"2026-09-15T23:37:05Z","doi":"10.1016/j.isci.2024.110030","publication_status":"peer_reviewed","year":2024,"artifact_sha256":"0183b6a111b1b02344cad35a571a1fd2c56257e406c5be1df69f7902c5d06749","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11167433/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:06Z","legacy_paper":{"id":"enhancer-position-encoding-2024","title":"A deep learning model for DNA enhancer prediction based on nucleotide position aware feature encoding","year":2024,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11167433/","primary_domain":"dna-genomes","retrieved_utc":"2026-09-15T23:37:05Z","notes":"Primary full text verified via Europe PMC fullTextXML; venue: iScience; PMC ID: PMC11167433. Task-specific CNN baseline, included as a DNA benchmark protocol reference.","doi":"10.1016/j.isci.2024.110030"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"ensemble-idp-docking-2025","kind":"source","name":"Ensemble docking for intrinsically disordered proteins","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11785235/","version":"preprint archived 2025-01-26","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.1101/2025.01.23.634614","publication_status":"preprint","year":2025,"artifact_sha256":"d02d91cdde41cb76ec5c86b532dffc564879c69e764a8c6b7752460fbbfd24b7","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11785235/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:56.275Z","legacy_paper":{"id":"ensemble-idp-docking-2025","title":"Ensemble docking for intrinsically disordered proteins","year":2025,"publication_status":"preprint","version":"preprint archived 2025-01-26","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11785235/","primary_domain":"molecular-interactions","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: bioRxiv; PMC ID: PMC11785235.","doi":"10.1101/2025.01.23.634614"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"ernie-rna-2025","kind":"source","name":"ERNIE-RNA: an RNA language model with structure-enhanced representations","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["rna-transcriptomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12627772/","version":"journal full text in PMC","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.1038/s41467-025-64972-0","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"0bd1d4b3cbf5d59d452cec4864614947861efcee050ba07e7de395cd90630047","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12627772/fullTextXML","artifact_retrieved_at":"2026-09-16T10:38:57.558206+00:00","legacy_paper":{"id":"ernie-rna-2025","title":"ERNIE-RNA: an RNA language model with structure-enhanced representations","year":2025,"publication_status":"peer_reviewed","version":"journal full text in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12627772/","primary_domain":"rna-transcriptomes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.1038/s41467-025-64972-0","notes":"Numeric result checked against Table 2 in primary full-text XML; journal/source: Nature Communications."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"esm2-amp-2025","kind":"source","name":"ESM2_AMP: an interpretable framework for protein–protein interactions prediction and biological mechanism discovery","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12392411/","version":"version of record","retrieved_at":"2026-09-15T23:37:05Z","doi":"10.1093/bib/bbaf434","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"8e7ad6efb72ca28d73037cdf465b0e62f99cd6d0ee4ca9eaf96a4c48da22fd6c","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12392411/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:58.625Z","legacy_paper":{"id":"esm2-amp-2025","title":"ESM2_AMP: an interpretable framework for protein–protein interactions prediction and biological mechanism discovery","year":2025,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12392411/","primary_domain":"molecular-interactions","retrieved_utc":"2026-09-15T23:37:05Z","notes":"Primary full text verified via Europe PMC fullTextXML; venue: Briefings in Bioinformatics; PMC ID: PMC12392411. Paper has multiple model variants; selected named ESM2_AMPS variant only.","doi":"10.1093/bib/bbaf434"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"esm2-ofs-fitness-2025","kind":"source","name":"Pseudo-perplexity in One Fell Swoop for Protein Fitness Estimation","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["proteins-complexes"]},"source_ids":[],"links":[],"attributes":{"url":"https://journals.aps.org/prxlife/pdf/10.1103/zhx7-hcmm","version":"PRX Life 2025 journal article","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.1103/zhx7-hcmm","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"085ef646f11b8e5335c4b3d86b15fb6c7bf5edf4a80a8b622753ac69d9991a67","artifact_url":"https://harvest.aps.org/v2/journals/articles/10.1103/zhx7-hcmm/fulltext","artifact_retrieved_at":"2026-09-16T10:45:41.099916+00:00","legacy_paper":{"id":"esm2-ofs-fitness-2025","title":"Pseudo-perplexity in One Fell Swoop for Protein Fitness Estimation","year":2025,"publication_status":"peer_reviewed","version":"PRX Life 2025 journal article","source_url":"https://journals.aps.org/prxlife/pdf/10.1103/zhx7-hcmm","primary_domain":"proteins-complexes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.1103/zhx7-hcmm","notes":"Final journal Table I, ESM2: OFS PP Aggregate Mean 0.403 checked directly; manuscript PMC11257618 printed the same value. Other models in the table are imported ProteinGym baselines; this row is the authors’ own evaluation."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-2ome-lm-2025","kind":"evaluation","name":"2OMe-LM: human RNA 2-prime-O-methylation site prediction","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["human RNA 2-prime-O-methylation site prediction"]},"source_ids":["2ome-lm-2025"],"links":[{"relation":"model","target_id":"reported-model-7f5b8234967c54"},{"relation":"benchmark","target_id":"reported-task-82fc7843f07324"},{"relation":"dataset","target_id":"reported-dataset-bd3d8e7d6cd196"}],"attributes":{"origin":"author_reported","protocol":"pretrained RNA language model predictor","version":"not stated in table","comparison":{"protocol_id":null,"dataset_version":null,"split":"5-fold cross-validation","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-antibody-deamidation-plm-2024","kind":"evaluation","name":"ESM-2 650M embeddings + classifier: antibody deamidation-site prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["antibody deamidation-site prediction"]},"source_ids":["antibody-deamidation-plm-2024"],"links":[{"relation":"model","target_id":"reported-model-4921459942b45f"},{"relation":"benchmark","target_id":"reported-task-0647b0364def8f"},{"relation":"dataset","target_id":"reported-dataset-0edd8f724db696"}],"attributes":{"origin":"author_reported","protocol":"global contextual embeddings only","version":"esm2_t33_650m_UR50D","comparison":{"protocol_id":null,"dataset_version":null,"split":"fivefold stratified CV","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-barcodebert-2026","kind":"evaluation","name":"BarcodeBERT (4–4-4): unseen-species genus classification","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["unseen-species genus classification"]},"source_ids":["barcodebert-2026"],"links":[{"relation":"model","target_id":"reported-model-05103f72325fe5"},{"relation":"benchmark","target_id":"reported-task-4a54ce01b5a855"},{"relation":"dataset","target_id":"reported-dataset-bc127dc9c441fe"}],"attributes":{"origin":"author_reported","protocol":"genus-level nearest-neighbor probe on species unseen in training","version":"4–4–4","comparison":{"protocol_id":null,"dataset_version":null,"split":"1-NN probe","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-birna-bert-2025","kind":"evaluation","name":"BiRNA-BERT: extremely long RNA species classification","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["extremely long RNA species classification"]},"source_ids":["birna-bert-2025"],"links":[{"relation":"model","target_id":"reported-model-d3fd83835a2d44"},{"relation":"benchmark","target_id":"reported-task-c40dac20d9af66"},{"relation":"dataset","target_id":"reported-dataset-ebc3f5fda43972"}],"attributes":{"origin":"author_reported","protocol":"adaptive tokenization on full-length long RNA sequences","version":"not stated in table","comparison":{"protocol_id":null,"dataset_version":null,"split":"paper evaluation","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-cathe2-2025","kind":"evaluation","name":"CATHe2 + ProstT5: CATH superfamily annotation","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["CATH superfamily annotation"]},"source_ids":["cathe2-2025"],"links":[{"relation":"model","target_id":"reported-model-49bc768f46b366"},{"relation":"benchmark","target_id":"reported-task-c98e91ffc7247d"},{"relation":"dataset","target_id":"reported-dataset-6e0c28dfde7337"}],"attributes":{"origin":"author_reported","protocol":"amino-acid and structural alphabet embedding classifier","version":"full ProstT5","comparison":{"protocol_id":null,"dataset_version":null,"split":"paper evaluation","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-clathrin-plm-2025","kind":"evaluation","name":"ESM-2 embedding + paper classifier: clathrin protein classification","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["clathrin protein classification"]},"source_ids":["clathrin-plm-2025"],"links":[{"relation":"model","target_id":"reported-model-e4710b1c3facf2"},{"relation":"benchmark","target_id":"reported-task-786c09824e9bf5"},{"relation":"dataset","target_id":"reported-dataset-0aab382ca2c063"}],"attributes":{"origin":"independent_paper","protocol":"single-feature ESM-2 embedding comparison","version":"not stated in table","comparison":{"protocol_id":null,"dataset_version":null,"split":"independent test","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-cobra-rna-binding-2026","kind":"evaluation","name":"ERNIE-RNA + CoBRA: RNA compound-binding site prediction","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA compound-binding site prediction"]},"source_ids":["cobra-rna-binding-2026"],"links":[{"relation":"model","target_id":"reported-model-7ad28cd57f5f5b"},{"relation":"benchmark","target_id":"reported-task-3a3bff34cce634"},{"relation":"dataset","target_id":"reported-dataset-b1af840b76b351"}],"attributes":{"origin":"author_reported","protocol":"ERNIE-RNA embedding with TCL focal loss","version":"not stated in table","comparison":{"protocol_id":null,"dataset_version":null,"split":"test set","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-codonbert-vaccines-2024","kind":"evaluation","name":"CodonBERT: flu-vaccine mRNA property prediction","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["flu-vaccine mRNA property prediction"]},"source_ids":["codonbert-vaccines-2024"],"links":[{"relation":"model","target_id":"reported-model-cd246741c378db"},{"relation":"benchmark","target_id":"reported-task-1c74661df2c401"},{"relation":"dataset","target_id":"reported-dataset-54b9bc432928d6"}],"attributes":{"origin":"author_reported","protocol":"codon-based model fine-tuned for downstream regression","version":"not stated in table","comparison":{"protocol_id":null,"dataset_version":null,"split":"paper evaluation","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-dart-eval-regulatory-2024","kind":"evaluation","name":"DNABERT-2: regulatory element identification","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["regulatory element identification"]},"source_ids":["dart-eval-regulatory-2024"],"links":[{"relation":"model","target_id":"reported-model-28413ae1766316"},{"relation":"benchmark","target_id":"reported-task-cdbee1c9285568"},{"relation":"dataset","target_id":"reported-dataset-b6ce37ba678d39"}],"attributes":{"origin":"independent_paper","protocol":"zero-shot likelihood ranking: higher likelihood for cCRE than matched control","version":"not stated in table","comparison":{"protocol_id":null,"dataset_version":null,"split":"paper evaluation","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-dnabert2-enhancer-2025","kind":"evaluation","name":"DNABERT2-Enhancer: enhancer recognition","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["enhancer recognition"]},"source_ids":["dnabert2-enhancer-2025"],"links":[{"relation":"model","target_id":"reported-model-d0e594ec3c0430"},{"relation":"benchmark","target_id":"reported-task-86a628af87ff8f"},{"relation":"dataset","target_id":"reported-dataset-6212e779949708"}],"attributes":{"origin":"author_reported","protocol":"first-layer enhancer versus non-enhancer classifier","version":"not stated in table","comparison":{"protocol_id":null,"dataset_version":null,"split":"5-fold cross-validation","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-eden-genomic-classification-2026","kind":"evaluation","name":"DNABERT-2: human core-promoter classification","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["human core-promoter classification"]},"source_ids":["eden-genomic-classification-2026"],"links":[{"relation":"model","target_id":"reported-model-2cb8118b4c77c0"},{"relation":"benchmark","target_id":"reported-task-9f62e739c6371e"},{"relation":"dataset","target_id":"reported-dataset-8e9488896becd4"}],"attributes":{"origin":"paper_compilation","protocol":"DNABERT-2 comparator in consolidated H-CPD table; rerun provenance not explicit","version":"not stated in table","comparison":{"protocol_id":null,"dataset_version":null,"split":"paper evaluation","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","original_evaluation":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-ernie-rna-2025","kind":"evaluation","name":"ERNIE-RNA: RNA secondary-structure prediction","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary-structure prediction"]},"source_ids":["ernie-rna-2025"],"links":[{"relation":"model","target_id":"reported-model-d023fbe78bc4df"},{"relation":"benchmark","target_id":"reported-task-a2bf7ddbc71d23"},{"relation":"dataset","target_id":"reported-dataset-abdfba8cce7486"}],"attributes":{"origin":"author_reported","protocol":"zero-shot attention-derived base-pair prediction","version":"86M","comparison":{"protocol_id":null,"dataset_version":null,"split":"cross-family test","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-esm2-ofs-fitness-2025","kind":"evaluation","name":"ESM2 OFS pseudo-perplexity: protein variant fitness prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["protein variant fitness prediction"]},"source_ids":["esm2-ofs-fitness-2025"],"links":[{"relation":"model","target_id":"reported-model-40ce004270dee4"},{"relation":"benchmark","target_id":"reported-task-c7a8a372f77886"},{"relation":"dataset","target_id":"reported-dataset-9c186c8f4ed3f4"}],"attributes":{"origin":"author_reported","protocol":"authors’ zero-shot ESM2 OFS pseudo-perplexity evaluation; aggregate mean across ProteinGym substitution assays","version":"not stated in table","comparison":{"protocol_id":null,"dataset_version":null,"split":"aggregate across assays","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-fusion-breakpoint-foundation-models-2026","kind":"evaluation","name":"Nucleotide Transformer + NN (middle): gene fusion breakpoint classification","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["gene fusion breakpoint classification"]},"source_ids":["fusion-breakpoint-foundation-models-2026"],"links":[{"relation":"model","target_id":"reported-model-1a67087ac262c5"},{"relation":"benchmark","target_id":"reported-task-ee34721cf55590"},{"relation":"dataset","target_id":"reported-dataset-f6922a9744ba27"}],"attributes":{"origin":"independent_paper","protocol":"middle embedding with neural-network classifier","version":"not stated in table","comparison":{"protocol_id":null,"dataset_version":null,"split":"full test set","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-genomic-tokenizer-selection-2025","kind":"evaluation","name":"Caduceus (character tokens): regulatory sequence classification","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["regulatory sequence classification"]},"source_ids":["genomic-tokenizer-selection-2025"],"links":[{"relation":"model","target_id":"reported-model-d5bc536ca6f0d3"},{"relation":"benchmark","target_id":"reported-task-cd127e56fb1f04"},{"relation":"dataset","target_id":"reported-dataset-0bba1c9a7ae410"}],"attributes":{"origin":"independent_paper","protocol":"task-category MCC across benchmark datasets","version":"3.9M parameter variant","comparison":{"protocol_id":null,"dataset_version":null,"split":"paper benchmark summary","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-gsmformer-ppi-2026","kind":"evaluation","name":"GSMFormer-PPI + ProstT5: protein-protein interaction prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["protein-protein interaction prediction"]},"source_ids":["gsmformer-ppi-2026"],"links":[{"relation":"model","target_id":"reported-model-73ae07fb5be204"},{"relation":"benchmark","target_id":"reported-task-dfa8f2285dbfa5"},{"relation":"dataset","target_id":"reported-dataset-07d355c146be1f"}],"attributes":{"origin":"author_reported","protocol":"ProstT5 embeddings as graph node features","version":"not stated in table","comparison":{"protocol_id":null,"dataset_version":null,"split":"test set","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-megsite-2025","kind":"evaluation","name":"MegSite + ESM3: DNA-binding residue prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["DNA-binding residue prediction"]},"source_ids":["megsite-2025"],"links":[{"relation":"model","target_id":"reported-model-86393c76dd8fa9"},{"relation":"benchmark","target_id":"reported-task-9917a0e69f33e7"},{"relation":"dataset","target_id":"reported-dataset-739aee3cf8d6f1"}],"attributes":{"origin":"author_reported","protocol":"ESM3 multimodal embedding ablation in MegSite","version":"not stated in table","comparison":{"protocol_id":null,"dataset_version":null,"split":"independent test","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-mrna-lm-2025","kind":"evaluation","name":"mRNA-LM: mRNA half-life prediction","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["mRNA half-life prediction"]},"source_ids":["mrna-lm-2025"],"links":[{"relation":"model","target_id":"reported-model-54d974e8e08043"},{"relation":"benchmark","target_id":"reported-task-5693847493f19f"},{"relation":"dataset","target_id":"reported-dataset-52f00ccaabf0d9"}],"attributes":{"origin":"author_reported","protocol":"average test performance across cross-validation splits","version":"not stated in table","comparison":{"protocol_id":null,"dataset_version":null,"split":"test set across CV splits","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-mrnabert-2025","kind":"evaluation","name":"mRNABERT: translation-efficiency prediction","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["translation-efficiency prediction"]},"source_ids":["mrnabert-2025"],"links":[{"relation":"model","target_id":"reported-model-13bd2a6c2d8178"},{"relation":"benchmark","target_id":"reported-task-f7142c3b3e0f3c"},{"relation":"dataset","target_id":"reported-dataset-1744719eef145b"}],"attributes":{"origin":"author_reported","protocol":"human translation-efficiency regression at 3066-nt input","version":"3066-nt input","comparison":{"protocol_id":null,"dataset_version":null,"split":"paper evaluation","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-mulan-2025","kind":"evaluation","name":"MULAN-ESM2 S: human protein-protein interaction prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["human protein-protein interaction prediction"]},"source_ids":["mulan-2025"],"links":[{"relation":"model","target_id":"reported-model-a29203c09857ef"},{"relation":"benchmark","target_id":"reported-task-6e54c7452b2b81"},{"relation":"dataset","target_id":"reported-dataset-38151fa548e291"}],"attributes":{"origin":"author_reported","protocol":"MULAN sequence-structure model based on ESM2 8M","version":"small ESM2 backbone","comparison":{"protocol_id":null,"dataset_version":null,"split":"paper evaluation","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-phylogpn-2025","kind":"evaluation","name":"PhyloGPN: ClinVar 3-prime UTR variant classification","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["ClinVar 3-prime UTR variant classification"]},"source_ids":["phylogpn-2025"],"links":[{"relation":"model","target_id":"reported-model-cbbe04b826ceff"},{"relation":"benchmark","target_id":"reported-task-ed3dd3b83c4505"},{"relation":"dataset","target_id":"reported-dataset-a28180d33f7a23"}],"attributes":{"origin":"author_reported","protocol":"log-likelihood-ratio scoring","version":"not stated in table","comparison":{"protocol_id":null,"dataset_version":null,"split":"paper evaluation","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-polya-glm-2025","kind":"evaluation","name":"HyenaDNA: polyadenylation site detection","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["polyadenylation site detection"]},"source_ids":["polya-glm-2025"],"links":[{"relation":"model","target_id":"reported-model-953007693fb72a"},{"relation":"benchmark","target_id":"reported-task-13dfe6b33e71ed"},{"relation":"dataset","target_id":"reported-dataset-55f200c9481409"}],"attributes":{"origin":"independent_paper","protocol":"few-shot Gene-Gene negative-set comparison","version":"not stated in table","comparison":{"protocol_id":null,"dataset_version":null,"split":"5-fold cross-validation","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-rlsite-rna-binding-2025","kind":"evaluation","name":"RLsite: RNA-small-molecule binding-site prediction","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA-small-molecule binding-site prediction"]},"source_ids":["rlsite-rna-binding-2025"],"links":[{"relation":"model","target_id":"reported-model-52b9c685d99290"},{"relation":"benchmark","target_id":"reported-task-b00a636d1ed8d9"},{"relation":"dataset","target_id":"reported-dataset-1c7f8ebb1968d9"}],"attributes":{"origin":"author_reported","protocol":"RNA language-model plus graph-attention classifier","version":"not stated in table","comparison":{"protocol_id":null,"dataset_version":null,"split":"paper evaluation","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-rnaret-2026","kind":"evaluation","name":"RNAret: miRNA-mRNA interaction prediction","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["miRNA-mRNA interaction prediction"]},"source_ids":["rnaret-2026"],"links":[{"relation":"model","target_id":"reported-model-7234658bc9c828"},{"relation":"benchmark","target_id":"reported-task-46e927bea10702"},{"relation":"dataset","target_id":"reported-dataset-99afd0c86b2954"}],"attributes":{"origin":"author_reported","protocol":"5-mer RNAret classifier; 72/8/20 train/validation/test split","version":"5-mer","comparison":{"protocol_id":null,"dataset_version":null,"split":"held-out test","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-spin-protein-function-2026","kind":"evaluation","name":"SPIN + ESM2-35M: protein function annotation","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["protein function annotation"]},"source_ids":["spin-protein-function-2026"],"links":[{"relation":"model","target_id":"reported-model-f8f0257b98749a"},{"relation":"benchmark","target_id":"reported-task-c4a578065f44b2"},{"relation":"dataset","target_id":"reported-dataset-dba1707164d296"}],"attributes":{"origin":"author_reported","protocol":"frozen ESM2-35M backbone in SPIN","version":"ESM2-35M frozen","comparison":{"protocol_id":null,"dataset_version":null,"split":"test set","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-structure-informed-plm-2025","kind":"evaluation","name":"structure-informed pLM: protein variant-effect classification","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["protein variant-effect classification"]},"source_ids":["structure-informed-plm-2025"],"links":[{"relation":"model","target_id":"reported-model-035a3ab36a3a6a"},{"relation":"benchmark","target_id":"reported-task-83be0998084c91"},{"relation":"dataset","target_id":"reported-dataset-2eaa2a051d45ee"}],"attributes":{"origin":"author_reported","protocol":"combined amino-acid, secondary structure, solvent accessibility and contact-map scoring","version":"not stated in table","comparison":{"protocol_id":null,"dataset_version":null,"split":"paper evaluation","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-001","kind":"evaluation","name":"Caduceus-Ph: Human 5mC detection","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["Human 5mC detection"]},"source_ids":["dna-foundation-models-2025"],"links":[{"relation":"model","target_id":"reported-model-47521865af7b04"},{"relation":"benchmark","target_id":"reported-task-988ff78f86471e"},{"relation":"dataset","target_id":"reported-dataset-463197d6a98b99"}],"attributes":{"origin":"independent_paper","protocol":"Binary epigenetic-modification classification as reported in the paper.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-002","kind":"evaluation","name":"NT-v2: Human 5mC detection","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["Human 5mC detection"]},"source_ids":["dna-foundation-models-2025"],"links":[{"relation":"model","target_id":"reported-model-3af86cb274f658"},{"relation":"benchmark","target_id":"reported-task-988ff78f86471e"},{"relation":"dataset","target_id":"reported-dataset-463197d6a98b99"}],"attributes":{"origin":"independent_paper","protocol":"Binary epigenetic-modification classification as reported in the paper.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-003","kind":"evaluation","name":"ENBED: Enhancer classification","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["Enhancer classification"]},"source_ids":["enbed-2024"],"links":[{"relation":"model","target_id":"reported-model-8db190bee6aae5"},{"relation":"benchmark","target_id":"reported-task-132da895d4c381"},{"relation":"dataset","target_id":"reported-dataset-f0bf60a62ad7c3"}],"attributes":{"origin":"author_reported","protocol":"Reported Genomic Benchmarks classification accuracy.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-004","kind":"evaluation","name":"ENBED (GRCh38): Enhancer classification","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["Enhancer classification"]},"source_ids":["enbed-2024"],"links":[{"relation":"model","target_id":"reported-model-bdb1db16d3389d"},{"relation":"benchmark","target_id":"reported-task-132da895d4c381"},{"relation":"dataset","target_id":"reported-dataset-f0bf60a62ad7c3"}],"attributes":{"origin":"author_reported","protocol":"ENBED trained on GRCh38; reported Genomic Benchmarks classification accuracy.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-005","kind":"evaluation","name":"DNABERT-2: G-quadruplex classification","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["G-quadruplex classification"]},"source_ids":["quadruplex-llm-benchmark-2025"],"links":[{"relation":"model","target_id":"reported-model-ade36035f58f27"},{"relation":"benchmark","target_id":"reported-task-c9d2a6435979e9"},{"relation":"dataset","target_id":"reported-dataset-9e9d18bc5bfb8b"}],"attributes":{"origin":"independent_paper","protocol":"Pretrained model evaluated on KEx as reported in Table 5.","version":"117M","comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-006","kind":"evaluation","name":"Caduceus: G-quadruplex classification","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["G-quadruplex classification"]},"source_ids":["quadruplex-llm-benchmark-2025"],"links":[{"relation":"model","target_id":"reported-model-abc19288fe9009"},{"relation":"benchmark","target_id":"reported-task-c9d2a6435979e9"},{"relation":"dataset","target_id":"reported-dataset-9e9d18bc5bfb8b"}],"attributes":{"origin":"independent_paper","protocol":"Pretrained model evaluated on KEx as reported in Table 5.","version":"8M","comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-007","kind":"evaluation","name":"HyenaDNA: Enhancer-target gene prediction","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["Enhancer-target gene prediction"]},"source_ids":["dnalongbench-2025"],"links":[{"relation":"model","target_id":"reported-model-9b3bc255532dd3"},{"relation":"benchmark","target_id":"reported-task-2cbac97dd849f5"},{"relation":"dataset","target_id":"reported-dataset-fcb5752d916d5d"}],"attributes":{"origin":"independent_paper","protocol":"Long-range ETGP benchmark; source table reports AUROC.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-008","kind":"evaluation","name":"Caduceus-Ph: Enhancer-target gene prediction","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["Enhancer-target gene prediction"]},"source_ids":["dnalongbench-2025"],"links":[{"relation":"model","target_id":"reported-model-a7cfacf25d97ad"},{"relation":"benchmark","target_id":"reported-task-2cbac97dd849f5"},{"relation":"dataset","target_id":"reported-dataset-fcb5752d916d5d"}],"attributes":{"origin":"independent_paper","protocol":"Long-range ETGP benchmark; source table reports AUROC.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-009","kind":"evaluation","name":"RiNALMo: Mean ribosome load from MPRA","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["Mean ribosome load from MPRA"]},"source_ids":["mrnabench-2025"],"links":[{"relation":"model","target_id":"reported-model-3e58d0faf88d2e"},{"relation":"benchmark","target_id":"reported-task-57dc3dcdb67a81"},{"relation":"dataset","target_id":"reported-dataset-3a3e3880a3fed0"}],"attributes":{"origin":"independent_paper","protocol":"Linear probe; mean across ten random seeds.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-010","kind":"evaluation","name":"RNA-FM: Mean ribosome load from MPRA","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["Mean ribosome load from MPRA"]},"source_ids":["mrnabench-2025"],"links":[{"relation":"model","target_id":"reported-model-43cf51abca83d1"},{"relation":"benchmark","target_id":"reported-task-57dc3dcdb67a81"},{"relation":"dataset","target_id":"reported-dataset-3a3e3880a3fed0"}],"attributes":{"origin":"independent_paper","protocol":"Linear probe; mean across ten random seeds.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-011","kind":"evaluation","name":"BPfold: RNA secondary structure","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary structure"]},"source_ids":["bpfold-2025"],"links":[{"relation":"model","target_id":"reported-model-c464bface507ee"},{"relation":"benchmark","target_id":"reported-task-dc82fcbfb44935"},{"relation":"dataset","target_id":"reported-dataset-8317793f18b026"}],"attributes":{"origin":"author_reported","protocol":"Family-wise evaluation of canonical base-pair predictions.","version":null,"comparison":{"protocol_id":null,"dataset_version":"116 RNAs","split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-012","kind":"evaluation","name":"RNAfold: RNA secondary structure","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary structure"]},"source_ids":["bpfold-2025"],"links":[{"relation":"model","target_id":"reported-model-52eee4cc67ca26"},{"relation":"benchmark","target_id":"reported-task-dc82fcbfb44935"},{"relation":"dataset","target_id":"reported-dataset-8317793f18b026"}],"attributes":{"origin":"independent_paper","protocol":"Family-wise evaluation of canonical base-pair predictions.","version":null,"comparison":{"protocol_id":null,"dataset_version":"116 RNAs","split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-013","kind":"evaluation","name":"TU-Fold (aug): RNA secondary structure","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary structure"]},"source_ids":["tu-fold-2025"],"links":[{"relation":"model","target_id":"reported-model-d1cd9a425f9bbd"},{"relation":"benchmark","target_id":"reported-task-5ec7581b246ea6"},{"relation":"dataset","target_id":"reported-dataset-f2e729f333a333"}],"attributes":{"origin":"author_reported","protocol":"Three-fold training and evaluation; source reports mean and standard deviation.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-014","kind":"evaluation","name":"UFold: RNA secondary structure","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary structure"]},"source_ids":["tu-fold-2025"],"links":[{"relation":"model","target_id":"reported-model-e3abb0b9a2ec79"},{"relation":"benchmark","target_id":"reported-task-5ec7581b246ea6"},{"relation":"dataset","target_id":"reported-dataset-f2e729f333a333"}],"attributes":{"origin":"independent_paper","protocol":"Three-fold training and evaluation; source reports mean and standard deviation.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-015","kind":"evaluation","name":"DEBFold: RNA secondary structure","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary structure"]},"source_ids":["debfold-2024"],"links":[{"relation":"model","target_id":"reported-model-3f850c08d76410"},{"relation":"benchmark","target_id":"reported-task-016f70615f2cfc"},{"relation":"dataset","target_id":"reported-dataset-18ebde58579c2b"}],"attributes":{"origin":"author_reported","protocol":"Median F1 on the prepared TestSetβ.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-016","kind":"evaluation","name":"RNAfold: RNA secondary structure","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary structure"]},"source_ids":["debfold-2024"],"links":[{"relation":"model","target_id":"reported-model-ed7f0db85facb1"},{"relation":"benchmark","target_id":"reported-task-016f70615f2cfc"},{"relation":"dataset","target_id":"reported-dataset-18ebde58579c2b"}],"attributes":{"origin":"independent_paper","protocol":"Median F1 on the prepared TestSetβ.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-017","kind":"evaluation","name":"ESM-2: Zero-shot substitution mutation effects: stability","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["Zero-shot substitution mutation effects: stability"]},"source_ids":["proteingym-2023"],"links":[{"relation":"model","target_id":"reported-model-d326e3c4e3ba20"},{"relation":"benchmark","target_id":"reported-task-6243658a1bc215"},{"relation":"dataset","target_id":"reported-dataset-7cec655cd742f3"}],"attributes":{"origin":"independent_paper","protocol":"Zero-shot mutation scores; average Spearman across stability-category assays.","version":"15B","comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-018","kind":"evaluation","name":"ProteinMPNN: Zero-shot substitution mutation effects: stability","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["Zero-shot substitution mutation effects: stability"]},"source_ids":["proteingym-2023"],"links":[{"relation":"model","target_id":"reported-model-e0443048c6e110"},{"relation":"benchmark","target_id":"reported-task-6243658a1bc215"},{"relation":"dataset","target_id":"reported-dataset-7cec655cd742f3"}],"attributes":{"origin":"independent_paper","protocol":"Zero-shot mutation scores; average Spearman across stability-category assays.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-019","kind":"evaluation","name":"FUJISAN: Enzyme functional identity prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["Enzyme functional identity prediction"]},"source_ids":["fujisan-2024"],"links":[{"relation":"model","target_id":"reported-model-9c10fbec02a365"},{"relation":"benchmark","target_id":"reported-task-1ebf9b408517f9"},{"relation":"dataset","target_id":"reported-dataset-5197cca532f89d"}],"attributes":{"origin":"author_reported","protocol":"Sequence and structural feature integration; paper-reported test sub-dataset.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-020","kind":"evaluation","name":"ESM2: Enzyme functional identity prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["Enzyme functional identity prediction"]},"source_ids":["fujisan-2024"],"links":[{"relation":"model","target_id":"reported-model-ccd1160ad4ec27"},{"relation":"benchmark","target_id":"reported-task-1ebf9b408517f9"},{"relation":"dataset","target_id":"reported-dataset-5197cca532f89d"}],"attributes":{"origin":"independent_paper","protocol":"Comparator evaluated on the paper-reported test sub-dataset.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-021","kind":"evaluation","name":"ESM-2: Mutated RBD binding prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["Mutated RBD binding prediction"]},"source_ids":["prime-2026"],"links":[{"relation":"model","target_id":"reported-model-d0d5df2beb02b2"},{"relation":"benchmark","target_id":"reported-task-00e594df6a182d"},{"relation":"dataset","target_id":"reported-dataset-becc215358afd0"}],"attributes":{"origin":"independent_paper","protocol":"Frozen mean-pooled representation with downstream regression; position-stratified split.","version":"8M","comparison":{"protocol_id":null,"dataset_version":null,"split":"position-stratified","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-022","kind":"evaluation","name":"ESM-C: Mutated RBD binding prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["Mutated RBD binding prediction"]},"source_ids":["prime-2026"],"links":[{"relation":"model","target_id":"reported-model-f83c0b833411a7"},{"relation":"benchmark","target_id":"reported-task-00e594df6a182d"},{"relation":"dataset","target_id":"reported-dataset-becc215358afd0"}],"attributes":{"origin":"independent_paper","protocol":"Frozen mean-pooled representation with downstream regression; position-stratified split.","version":"300M","comparison":{"protocol_id":null,"dataset_version":null,"split":"position-stratified","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-023","kind":"evaluation","name":"PST: Zero-shot variant effect prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["Zero-shot variant effect prediction"]},"source_ids":["pst-2025"],"links":[{"relation":"model","target_id":"reported-model-7dd5992188a868"},{"relation":"benchmark","target_id":"reported-task-a5141363b0ee45"},{"relation":"dataset","target_id":"reported-dataset-bd9255afb783d6"}],"attributes":{"origin":"author_reported","protocol":"Zero-shot VEP; paper averages absolute Spearman correlations.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-024","kind":"evaluation","name":"ESM-2: Zero-shot variant effect prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["Zero-shot variant effect prediction"]},"source_ids":["pst-2025"],"links":[{"relation":"model","target_id":"reported-model-d25dab1a9c4fff"},{"relation":"benchmark","target_id":"reported-task-a5141363b0ee45"},{"relation":"dataset","target_id":"reported-dataset-bd9255afb783d6"}],"attributes":{"origin":"independent_paper","protocol":"Zero-shot VEP; paper averages absolute Spearman correlations.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-025","kind":"evaluation","name":"scGPT: Cell-type identification","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Cell-type identification"]},"source_ids":["single-cell-peft-2024"],"links":[{"relation":"model","target_id":"reported-model-3bdb3093e8d531"},{"relation":"benchmark","target_id":"reported-task-5b929593eefc76"},{"relation":"dataset","target_id":"reported-dataset-488d5de6bb9c1b"}],"attributes":{"origin":"independent_paper","protocol":"Native scLLM cell-type identification as reported in Table 2.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-026","kind":"evaluation","name":"Geneformer: Cell-type identification","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Cell-type identification"]},"source_ids":["single-cell-peft-2024"],"links":[{"relation":"model","target_id":"reported-model-b46ae14b9927ac"},{"relation":"benchmark","target_id":"reported-task-5b929593eefc76"},{"relation":"dataset","target_id":"reported-dataset-488d5de6bb9c1b"}],"attributes":{"origin":"independent_paper","protocol":"Native scLLM cell-type identification as reported in Table 2.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-027","kind":"evaluation","name":"C2S (GPT-2 Large): Combinatorial cell-label classification","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Combinatorial cell-label classification"]},"source_ids":["cell2sentence-2024"],"links":[{"relation":"model","target_id":"reported-model-ab02228f50a37c"},{"relation":"benchmark","target_id":"reported-task-7efe245cc94ee5"},{"relation":"dataset","target_id":"reported-dataset-87e91d9d6e6f4b"}],"attributes":{"origin":"author_reported","protocol":"Partial-credit labels including cell type, perturbation, and dose.","version":"GPT-2 Large","comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-028","kind":"evaluation","name":"Geneformer: Combinatorial cell-label classification","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Combinatorial cell-label classification"]},"source_ids":["cell2sentence-2024"],"links":[{"relation":"model","target_id":"reported-model-06816ce9073144"},{"relation":"benchmark","target_id":"reported-task-7efe245cc94ee5"},{"relation":"dataset","target_id":"reported-dataset-87e91d9d6e6f4b"}],"attributes":{"origin":"independent_paper","protocol":"Partial-credit labels including cell type, perturbation, and dose.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-029","kind":"evaluation","name":"scGPT: Cell-type annotation","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Cell-type annotation"]},"source_ids":["scelmo-2025"],"links":[{"relation":"model","target_id":"reported-model-77ad27d4098177"},{"relation":"benchmark","target_id":"reported-task-660753ec94e631"},{"relation":"dataset","target_id":"reported-dataset-8f123f006964ad"}],"attributes":{"origin":"paper_compilation","protocol":"Zero-shot setting; source caption says some comparator rows come from GenePT.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","original_evaluation":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-030","kind":"evaluation","name":"Geneformer: Cell-type annotation","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Cell-type annotation"]},"source_ids":["scelmo-2025"],"links":[{"relation":"model","target_id":"reported-model-d60f505aabb19c"},{"relation":"benchmark","target_id":"reported-task-660753ec94e631"},{"relation":"dataset","target_id":"reported-dataset-8f123f006964ad"}],"attributes":{"origin":"paper_compilation","protocol":"Zero-shot setting; source caption says some comparator rows come from GenePT.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","original_evaluation":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-031","kind":"evaluation","name":"scRegNet (Geneformer backbone): Gene-regulatory link prediction","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Gene-regulatory link prediction"]},"source_ids":["scregnet-2025"],"links":[{"relation":"model","target_id":"reported-model-60455ff7cc0c15"},{"relation":"benchmark","target_id":"reported-task-3063ed4da76b4b"},{"relation":"dataset","target_id":"reported-dataset-2ad2fad5e1cd0a"}],"attributes":{"origin":"author_reported","protocol":"TFs plus 500 variable genes; mean from 50 independent evaluations.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-032","kind":"evaluation","name":"scRegNet (scBERT backbone): Gene-regulatory link prediction","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Gene-regulatory link prediction"]},"source_ids":["scregnet-2025"],"links":[{"relation":"model","target_id":"reported-model-89f5a8f309fa18"},{"relation":"benchmark","target_id":"reported-task-3063ed4da76b4b"},{"relation":"dataset","target_id":"reported-dataset-2ad2fad5e1cd0a"}],"attributes":{"origin":"author_reported","protocol":"TFs plus 500 variable genes; mean from 50 independent evaluations.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-033","kind":"evaluation","name":"ProkBERT-mini: E. coli sigma70 promoter prediction","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["E. coli sigma70 promoter prediction"]},"source_ids":["prokbert-2024"],"links":[{"relation":"model","target_id":"reported-model-4438513d9cd42c"},{"relation":"benchmark","target_id":"reported-task-3891811dcce8b3"},{"relation":"dataset","target_id":"reported-dataset-48def1da574597"}],"attributes":{"origin":"author_reported","protocol":"Promoter versus non-promoter classification.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-034","kind":"evaluation","name":"Promotech: E. coli sigma70 promoter prediction","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["E. coli sigma70 promoter prediction"]},"source_ids":["prokbert-2024"],"links":[{"relation":"model","target_id":"reported-model-0d147487bf97be"},{"relation":"benchmark","target_id":"reported-task-3891811dcce8b3"},{"relation":"dataset","target_id":"reported-dataset-48def1da574597"}],"attributes":{"origin":"independent_paper","protocol":"Promoter versus non-promoter classification.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-035","kind":"evaluation","name":"Eco70PromBERT: E. coli sigma70 promoter prediction","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["E. coli sigma70 promoter prediction"]},"source_ids":["cyaprombert-2022"],"links":[{"relation":"model","target_id":"reported-model-5b70fccb70bb70"},{"relation":"benchmark","target_id":"reported-task-e5c34f686ac403"},{"relation":"dataset","target_id":"reported-dataset-a1da4a37eb46a5"}],"attributes":{"origin":"author_reported","protocol":"BERT-base with 1bp tokenizer; 110 promoters and 108 non-promoters.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":"independent test","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-036","kind":"evaluation","name":"iPro70-FMWin: E. coli sigma70 promoter prediction","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["E. coli sigma70 promoter prediction"]},"source_ids":["cyaprombert-2022"],"links":[{"relation":"model","target_id":"reported-model-23cb15b93c00ff"},{"relation":"benchmark","target_id":"reported-task-e5c34f686ac403"},{"relation":"dataset","target_id":"reported-dataset-a1da4a37eb46a5"}],"attributes":{"origin":"independent_paper","protocol":"Compared on the same independent test dataset; 110 promoters and 108 non-promoters.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":"independent test","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-037","kind":"evaluation","name":"EVO2: Genome-wide prophage detection","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Genome-wide prophage detection"]},"source_ids":["lambda-prophage-2026"],"links":[{"relation":"model","target_id":"reported-model-aa763db2cfdeff"},{"relation":"benchmark","target_id":"reported-task-dd001540e0f4ec"},{"relation":"dataset","target_id":"reported-dataset-1b4f6ea24c0587"}],"attributes":{"origin":"independent_paper","protocol":"Genomic language model fine-tuned for prophage detection; genome-wide evaluation.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-038","kind":"evaluation","name":"geNomad: Genome-wide prophage detection","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Genome-wide prophage detection"]},"source_ids":["lambda-prophage-2026"],"links":[{"relation":"model","target_id":"reported-model-d0677d52d2b9fd"},{"relation":"benchmark","target_id":"reported-task-dd001540e0f4ec"},{"relation":"dataset","target_id":"reported-dataset-1b4f6ea24c0587"}],"attributes":{"origin":"independent_paper","protocol":"Traditional specialist comparator; genome-wide evaluation.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-039","kind":"evaluation","name":"NABAS+: Metagenomic taxonomic classification","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Metagenomic taxonomic classification"]},"source_ids":["nabas-plus-2025"],"links":[{"relation":"model","target_id":"reported-model-e7d203bd99ca99"},{"relation":"benchmark","target_id":"reported-task-92137759a9e7b0"},{"relation":"dataset","target_id":"reported-dataset-b462aa24561fba"}],"attributes":{"origin":"author_reported","protocol":"Newly generated sample19 used for classifier comparison.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-040","kind":"evaluation","name":"MetaPhlAn3: Metagenomic taxonomic classification","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Metagenomic taxonomic classification"]},"source_ids":["nabas-plus-2025"],"links":[{"relation":"model","target_id":"reported-model-df4084611520b7"},{"relation":"benchmark","target_id":"reported-task-92137759a9e7b0"},{"relation":"dataset","target_id":"reported-dataset-b462aa24561fba"}],"attributes":{"origin":"independent_paper","protocol":"Newly generated sample19 used for classifier comparison.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-041","kind":"evaluation","name":"Chai-1: Lipid–protein binding pose","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Lipid–protein binding pose"]},"source_ids":["lipp-2026"],"links":[{"relation":"model","target_id":"reported-model-eae60780097101"},{"relation":"benchmark","target_id":"reported-task-ff2dec63c5a3dd"},{"relation":"dataset","target_id":"reported-dataset-6a44f5946cd7ab"}],"attributes":{"origin":"independent_paper","protocol":"Top-scoring pose; all-atom lipid RMSD below 2 Å.","version":null,"comparison":{"protocol_id":null,"dataset_version":"331 complexes","split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-042","kind":"evaluation","name":"DiffDock-L: Lipid–protein binding pose","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Lipid–protein binding pose"]},"source_ids":["lipp-2026"],"links":[{"relation":"model","target_id":"reported-model-51ed86132346a0"},{"relation":"benchmark","target_id":"reported-task-ff2dec63c5a3dd"},{"relation":"dataset","target_id":"reported-dataset-6a44f5946cd7ab"}],"attributes":{"origin":"independent_paper","protocol":"Top-scoring pose; all-atom lipid RMSD below 2 Å.","version":null,"comparison":{"protocol_id":null,"dataset_version":"331 complexes","split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-043","kind":"evaluation","name":"DiffDock-NMDN: Protein–ligand virtual screening","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand virtual screening"]},"source_ids":["nmdn-2025"],"links":[{"relation":"model","target_id":"reported-model-6c0bc8d297cc7a"},{"relation":"benchmark","target_id":"reported-task-a7803ecf7708cc"},{"relation":"dataset","target_id":"reported-dataset-065b9fcc8da573"}],"attributes":{"origin":"author_reported","protocol":"NMDN scoring on DiffDock-NMDN blind docked poses; not ligand-pose RMSD.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-044","kind":"evaluation","name":"Vina: Protein–ligand virtual screening","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand virtual screening"]},"source_ids":["nmdn-2025"],"links":[{"relation":"model","target_id":"reported-model-1e51ccbfd2de61"},{"relation":"benchmark","target_id":"reported-task-a7803ecf7708cc"},{"relation":"dataset","target_id":"reported-dataset-065b9fcc8da573"}],"attributes":{"origin":"independent_paper","protocol":"Vina scoring on the same DiffDock-NMDN blind docked poses; not ligand-pose RMSD.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-045","kind":"evaluation","name":"Boltz-1: Protein–ligand pose prediction","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand pose prediction"]},"source_ids":["boltz-stereochemistry-2025"],"links":[{"relation":"model","target_id":"reported-model-d9a06805b36b8a"},{"relation":"benchmark","target_id":"reported-task-bf513ed6db92c5"},{"relation":"dataset","target_id":"reported-dataset-5afaefb87c8a94"}],"attributes":{"origin":"independent_paper","protocol":"All entries; authors note this dataset contains structures seen during model training.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-046","kind":"evaluation","name":"DiffDock: Protein–ligand pose prediction","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand pose prediction"]},"source_ids":["boltz-stereochemistry-2025"],"links":[{"relation":"model","target_id":"reported-model-7f6ffd9e2a08be"},{"relation":"benchmark","target_id":"reported-task-bf513ed6db92c5"},{"relation":"dataset","target_id":"reported-dataset-5afaefb87c8a94"}],"attributes":{"origin":"independent_paper","protocol":"All entries; rigid-protein docking comparator; authors note this dataset contains structures seen during model training.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-047","kind":"evaluation","name":"Boltz-2: Ligand potency prediction using generated poses","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Ligand potency prediction using generated poses"]},"source_ids":["mpro-pose-affinity-2025"],"links":[{"relation":"model","target_id":"reported-model-cdc9aabf4efc04"},{"relation":"benchmark","target_id":"reported-task-d5f897ab0f6f67"},{"relation":"dataset","target_id":"reported-dataset-235520c84b737f"}],"attributes":{"origin":"independent_paper","protocol":"Potency prediction using Boltz-2 ligand-pose generation protocol; see paper scoring pipeline.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-048","kind":"evaluation","name":"DiffDock: Ligand potency prediction using generated poses","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Ligand potency prediction using generated poses"]},"source_ids":["mpro-pose-affinity-2025"],"links":[{"relation":"model","target_id":"reported-model-415ee22f46526c"},{"relation":"benchmark","target_id":"reported-task-d5f897ab0f6f67"},{"relation":"dataset","target_id":"reported-dataset-235520c84b737f"}],"attributes":{"origin":"independent_paper","protocol":"Potency prediction using DiffDock ligand-pose generation plus paper scoring pipeline; not a native DiffDock affinity score.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-003","kind":"evaluation","name":"Mouse-Geneformer: Human thymus cell-type classification","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Human thymus cell-type classification"]},"source_ids":["mouse-geneformer-2025"],"links":[{"relation":"model","target_id":"reported-model-e2f2f0d4830bb0"},{"relation":"benchmark","target_id":"reported-task-031186b57c62de"},{"relation":"dataset","target_id":"reported-dataset-477a9082515406"}],"attributes":{"origin":"author_reported","protocol":"Ortholog-based gene conversion; zero-shot mouse model on human cells.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-004","kind":"evaluation","name":"Human-Geneformer: Human thymus cell-type classification","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Human thymus cell-type classification"]},"source_ids":["mouse-geneformer-2025"],"links":[{"relation":"model","target_id":"reported-model-10d85f2a035720"},{"relation":"benchmark","target_id":"reported-task-031186b57c62de"},{"relation":"dataset","target_id":"reported-dataset-477a9082515406"}],"attributes":{"origin":"independent_paper","protocol":"Native human model; zero-shot setting.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-005","kind":"evaluation","name":"scLLMDA: Cross-platform scATAC cell-type annotation","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Cross-platform scATAC cell-type annotation"]},"source_ids":["scatac-llmda-2026"],"links":[{"relation":"model","target_id":"reported-model-a0db32ae53e5ed"},{"relation":"benchmark","target_id":"reported-task-d82b6284f3f431"},{"relation":"dataset","target_id":"reported-dataset-7fc59ce4c0ceaa"}],"attributes":{"origin":"author_reported","protocol":"Cross-platform reference-query cell-type annotation.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-006","kind":"evaluation","name":"MINGLE: Cross-platform scATAC cell-type annotation","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Cross-platform scATAC cell-type annotation"]},"source_ids":["scatac-llmda-2026"],"links":[{"relation":"model","target_id":"reported-model-f0c630d0565e64"},{"relation":"benchmark","target_id":"reported-task-d82b6284f3f431"},{"relation":"dataset","target_id":"reported-dataset-7fc59ce4c0ceaa"}],"attributes":{"origin":"independent_paper","protocol":"Cross-platform reference-query comparator.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-011","kind":"evaluation","name":"GenePT-w: Cell-type structure in frozen embeddings","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Cell-type structure in frozen embeddings"]},"source_ids":["genept-2024"],"links":[{"relation":"model","target_id":"reported-model-7c595040de69bc"},{"relation":"benchmark","target_id":"reported-task-4df1fb456d3deb"},{"relation":"dataset","target_id":"reported-dataset-eaa2965545c87b"}],"attributes":{"origin":"author_reported","protocol":"k-means on pretrained cell embeddings; agreement with original cell-type labels.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-012","kind":"evaluation","name":"scGPT: Cell-type structure in frozen embeddings","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Cell-type structure in frozen embeddings"]},"source_ids":["genept-2024"],"links":[{"relation":"model","target_id":"reported-model-399b1ce87a3f6d"},{"relation":"benchmark","target_id":"reported-task-4df1fb456d3deb"},{"relation":"dataset","target_id":"reported-dataset-eaa2965545c87b"}],"attributes":{"origin":"independent_paper","protocol":"k-means on pretrained cell embeddings; agreement with original cell-type labels.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-013","kind":"evaluation","name":"Best frozen single-cell foundation model: Donor-aware age-class prediction","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Donor-aware age-class prediction"]},"source_ids":["single-cell-aging-probes-2026"],"links":[{"relation":"model","target_id":"reported-model-148b613975b6eb"},{"relation":"benchmark","target_id":"reported-task-d6018ca598e525"},{"relation":"dataset","target_id":"reported-dataset-571ce000cd74b9"}],"attributes":{"origin":"independent_paper","protocol":"Same donor-aware splits and logistic-regression probe as expression PCA; text names Geneformer as best model on AIDA v2.","version":null,"comparison":{"protocol_id":null,"dataset_version":"622 donors","split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-014","kind":"evaluation","name":"Gene-expression PCA: Donor-aware age-class prediction","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Donor-aware age-class prediction"]},"source_ids":["single-cell-aging-probes-2026"],"links":[{"relation":"model","target_id":"reported-model-177f32ce8189a0"},{"relation":"benchmark","target_id":"reported-task-d6018ca598e525"},{"relation":"dataset","target_id":"reported-dataset-571ce000cd74b9"}],"attributes":{"origin":"independent_paper","protocol":"Fifty-component gene-expression PCA with the same donor-aware probe splits.","version":null,"comparison":{"protocol_id":null,"dataset_version":"622 donors","split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-015","kind":"evaluation","name":"scaLR: PBMC cell-type classification","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["PBMC cell-type classification"]},"source_ids":["scalr-2025"],"links":[{"relation":"model","target_id":"reported-model-7cf2f9951e1dba"},{"relation":"benchmark","target_id":"reported-task-b46b7b839bff93"},{"relation":"dataset","target_id":"reported-dataset-33a41fe5fc66cf"}],"attributes":{"origin":"author_reported","protocol":"All features and samples from PBMCs-BS.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-016","kind":"evaluation","name":"scVI + scANVI: PBMC cell-type classification","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["PBMC cell-type classification"]},"source_ids":["scalr-2025"],"links":[{"relation":"model","target_id":"reported-model-1be5c4b7c52a41"},{"relation":"benchmark","target_id":"reported-task-b46b7b839bff93"},{"relation":"dataset","target_id":"reported-dataset-33a41fe5fc66cf"}],"attributes":{"origin":"independent_paper","protocol":"All features and samples from PBMCs-BS; comparison pipeline combines scVI and scANVI.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-017","kind":"evaluation","name":"scXDR: Cross-dataset single-cell drug response transfer","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Cross-dataset single-cell drug response transfer"]},"source_ids":["scxdr-2026"],"links":[{"relation":"model","target_id":"reported-model-9f39de53f7a139"},{"relation":"benchmark","target_id":"reported-task-167f08013c270e"},{"relation":"dataset","target_id":"reported-dataset-f7210686a78474"}],"attributes":{"origin":"author_reported","protocol":"Single-cell-to-single-cell transfer; source scenario 2.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-018","kind":"evaluation","name":"scVI: Cross-dataset single-cell drug response transfer","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Cross-dataset single-cell drug response transfer"]},"source_ids":["scxdr-2026"],"links":[{"relation":"model","target_id":"reported-model-f23306b94dc7b6"},{"relation":"benchmark","target_id":"reported-task-167f08013c270e"},{"relation":"dataset","target_id":"reported-dataset-f7210686a78474"}],"attributes":{"origin":"independent_paper","protocol":"Single-cell-to-single-cell transfer; source scenario 2.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-019","kind":"evaluation","name":"CAMMiQ: Strain-level abundance quantification","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Strain-level abundance quantification"]},"source_ids":["cammiq-2022"],"links":[{"relation":"model","target_id":"reported-model-32a19f43a4c254"},{"relation":"benchmark","target_id":"reported-task-571f0a2e7faed3"},{"relation":"dataset","target_id":"reported-dataset-72e837e5b97041"}],"attributes":{"origin":"author_reported","protocol":"Strain-level quantification on the HumanGut-all synthetic query.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-020","kind":"evaluation","name":"Kraken2: Strain-level abundance quantification","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Strain-level abundance quantification"]},"source_ids":["cammiq-2022"],"links":[{"relation":"model","target_id":"reported-model-70f57ebb163a5c"},{"relation":"benchmark","target_id":"reported-task-571f0a2e7faed3"},{"relation":"dataset","target_id":"reported-dataset-72e837e5b97041"}],"attributes":{"origin":"independent_paper","protocol":"Strain-level quantification on the HumanGut-all synthetic query.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-021","kind":"evaluation","name":"Lazypipe-nt: Simulated metagenome virus-taxon retrieval","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Simulated metagenome virus-taxon retrieval"]},"source_ids":["lazypipe-2020"],"links":[{"relation":"model","target_id":"reported-model-8cb3dd4e9f5b10"},{"relation":"benchmark","target_id":"reported-task-369dcfef14c4a9"},{"relation":"dataset","target_id":"reported-dataset-450c1af18cc623"}],"attributes":{"origin":"author_reported","protocol":"Genus-rank viral taxon retrieval.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-022","kind":"evaluation","name":"Kraken2: Simulated metagenome virus-taxon retrieval","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Simulated metagenome virus-taxon retrieval"]},"source_ids":["lazypipe-2020"],"links":[{"relation":"model","target_id":"reported-model-673b8f46361000"},{"relation":"benchmark","target_id":"reported-task-369dcfef14c4a9"},{"relation":"dataset","target_id":"reported-dataset-450c1af18cc623"}],"attributes":{"origin":"independent_paper","protocol":"Genus-rank viral taxon retrieval.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-023","kind":"evaluation","name":"NCD-gzip: CAMI II superkingdom read classification","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["CAMI II superkingdom read classification"]},"source_ids":["ncd-metagenomics-2026"],"links":[{"relation":"model","target_id":"reported-model-7b052acf17b5ba"},{"relation":"benchmark","target_id":"reported-task-a2c37b8c420bc3"},{"relation":"dataset","target_id":"reported-dataset-beb4f5da29da0a"}],"attributes":{"origin":"author_reported","protocol":"Superkingdom-level macro-averaged F1; NCD assigns every read.","version":null,"comparison":{"protocol_id":null,"dataset_version":"10,000 reads","split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-024","kind":"evaluation","name":"NCD-gzip: CAMI II phylum read classification","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["CAMI II phylum read classification"]},"source_ids":["ncd-metagenomics-2026"],"links":[{"relation":"model","target_id":"reported-model-7b052acf17b5ba"},{"relation":"benchmark","target_id":"reported-task-45105e1c486251"},{"relation":"dataset","target_id":"reported-dataset-beb4f5da29da0a"}],"attributes":{"origin":"author_reported","protocol":"Phylum-level macro-averaged F1; distinct taxonomic rank from the other row.","version":null,"comparison":{"protocol_id":null,"dataset_version":"10,000 reads","split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-025","kind":"evaluation","name":"VIBRANT: Simulated prophage-contig detection","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Simulated prophage-contig detection"]},"source_ids":["viral-contig-simulation-2021"],"links":[{"relation":"model","target_id":"reported-model-27dca28a87cf3c"},{"relation":"benchmark","target_id":"reported-task-53e3d216eef6db"},{"relation":"dataset","target_id":"reported-dataset-cd51026cdb6a7a"}],"attributes":{"origin":"independent_paper","protocol":"Average across twenty medium- and high-complexity simulated communities.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-026","kind":"evaluation","name":"VirSorter: Simulated prophage-contig detection","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Simulated prophage-contig detection"]},"source_ids":["viral-contig-simulation-2021"],"links":[{"relation":"model","target_id":"reported-model-e78e3886df0d3a"},{"relation":"benchmark","target_id":"reported-task-53e3d216eef6db"},{"relation":"dataset","target_id":"reported-dataset-cd51026cdb6a7a"}],"attributes":{"origin":"independent_paper","protocol":"Average across twenty medium- and high-complexity simulated communities.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-027","kind":"evaluation","name":"GenomeOcean: Natural vs artificial microbial genome sequence","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Natural vs artificial microbial genome sequence"]},"source_ids":["genomeocean-2025"],"links":[{"relation":"model","target_id":"reported-model-2df975e60d16d1"},{"relation":"benchmark","target_id":"reported-task-9f9ab0090f6522"},{"relation":"dataset","target_id":"reported-dataset-0c3ac7efe99c37"}],"attributes":{"origin":"author_reported","protocol":"Source reports natural-versus-artificial sequence classification.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-028","kind":"evaluation","name":"DNABERT-2: Natural vs artificial microbial genome sequence","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Natural vs artificial microbial genome sequence"]},"source_ids":["genomeocean-2025"],"links":[{"relation":"model","target_id":"reported-model-7f1165b35f10e2"},{"relation":"benchmark","target_id":"reported-task-9f9ab0090f6522"},{"relation":"dataset","target_id":"reported-dataset-0c3ac7efe99c37"}],"attributes":{"origin":"independent_paper","protocol":"Source reports natural-versus-artificial sequence classification.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-029","kind":"evaluation","name":"kMetaShot: Mock-community MAG taxonomy classification","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Mock-community MAG taxonomy classification"]},"source_ids":["kmetashot-2025"],"links":[{"relation":"model","target_id":"reported-model-790768ed581685"},{"relation":"benchmark","target_id":"reported-task-8406b6aabfb8c0"},{"relation":"dataset","target_id":"reported-dataset-8cad416ddc80dc"}],"attributes":{"origin":"author_reported","protocol":"Genus classification of MAGs from MegaHIT contigs; uncorrected kMetaShot.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-030","kind":"evaluation","name":"GTDB-Tk: Mock-community MAG taxonomy classification","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Mock-community MAG taxonomy classification"]},"source_ids":["kmetashot-2025"],"links":[{"relation":"model","target_id":"reported-model-3fd1e9f6c573b2"},{"relation":"benchmark","target_id":"reported-task-8406b6aabfb8c0"},{"relation":"dataset","target_id":"reported-dataset-8cad416ddc80dc"}],"attributes":{"origin":"independent_paper","protocol":"Genus classification of the same MAG set.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-031","kind":"evaluation","name":"Lemur: Long-read taxonomic profiling","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Long-read taxonomic profiling"]},"source_ids":["lemur-magnet-2024"],"links":[{"relation":"model","target_id":"reported-model-6ac0730e8481de"},{"relation":"benchmark","target_id":"reported-task-6330d593980b5b"},{"relation":"dataset","target_id":"reported-dataset-768a7ff5bac414"}],"attributes":{"origin":"author_reported","protocol":"Mean across five replicate runs on Zymo LOG 10%.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-032","kind":"evaluation","name":"Kraken 2: Long-read taxonomic profiling","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Long-read taxonomic profiling"]},"source_ids":["lemur-magnet-2024"],"links":[{"relation":"model","target_id":"reported-model-62bc5e5ba13e7d"},{"relation":"benchmark","target_id":"reported-task-6330d593980b5b"},{"relation":"dataset","target_id":"reported-dataset-768a7ff5bac414"}],"attributes":{"origin":"independent_paper","protocol":"Mean across five replicate runs on Zymo LOG 10%.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-033","kind":"evaluation","name":"iPro-MP: Multi-species prokaryotic promoter detection","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Multi-species prokaryotic promoter detection"]},"source_ids":["ipromp-2025"],"links":[{"relation":"model","target_id":"reported-model-9a200c55b0e03e"},{"relation":"benchmark","target_id":"reported-task-a1151e386a3d3f"},{"relation":"dataset","target_id":"reported-dataset-e0f34dcaa1ba3b"}],"attributes":{"origin":"author_reported","protocol":"Average over independent testing sets.","version":null,"comparison":{"protocol_id":null,"dataset_version":"23 test sets","split":"independent test","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-034","kind":"evaluation","name":"Prompt: Multi-species prokaryotic promoter detection","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Multi-species prokaryotic promoter detection"]},"source_ids":["ipromp-2025"],"links":[{"relation":"model","target_id":"reported-model-03080a5289c07e"},{"relation":"benchmark","target_id":"reported-task-a1151e386a3d3f"},{"relation":"dataset","target_id":"reported-dataset-e0f34dcaa1ba3b"}],"attributes":{"origin":"independent_paper","protocol":"Average over the same independent testing sets.","version":null,"comparison":{"protocol_id":null,"dataset_version":"23 test sets","split":"independent test","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-035","kind":"evaluation","name":"ICCTax: Hierarchical metagenomic taxonomy classification","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Hierarchical metagenomic taxonomy classification"]},"source_ids":["icctax-2025"],"links":[{"relation":"model","target_id":"reported-model-67eaf766fa9877"},{"relation":"benchmark","target_id":"reported-task-f4b1c9373f0929"},{"relation":"dataset","target_id":"reported-dataset-561834dfa1682c"}],"attributes":{"origin":"author_reported","protocol":"Macro average precision at genus rank on Complete dataset.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-036","kind":"evaluation","name":"Kraken2: Hierarchical metagenomic taxonomy classification","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Hierarchical metagenomic taxonomy classification"]},"source_ids":["icctax-2025"],"links":[{"relation":"model","target_id":"reported-model-8861b9ad9b9c9b"},{"relation":"benchmark","target_id":"reported-task-f4b1c9373f0929"},{"relation":"dataset","target_id":"reported-dataset-561834dfa1682c"}],"attributes":{"origin":"independent_paper","protocol":"Macro average precision at genus rank on Complete dataset.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-037","kind":"evaluation","name":"Chai-1: Antibody–antigen interaction prediction using folded complexes","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Antibody–antigen interaction prediction using folded complexes"]},"source_ids":["antibody-flexibility-2025"],"links":[{"relation":"model","target_id":"reported-model-b3fdf259d51533"},{"relation":"benchmark","target_id":"reported-task-0c92cda11228c4"},{"relation":"dataset","target_id":"reported-dataset-ee26acbd6e8cf7"}],"attributes":{"origin":"independent_paper","protocol":"Interaction classifier evaluated using Chai-1-folded input complexes; this is pipeline AUC, not DockQ.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-038","kind":"evaluation","name":"Boltz-1: Antibody–antigen interaction prediction using folded complexes","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Antibody–antigen interaction prediction using folded complexes"]},"source_ids":["antibody-flexibility-2025"],"links":[{"relation":"model","target_id":"reported-model-884582fb0c70dc"},{"relation":"benchmark","target_id":"reported-task-0c92cda11228c4"},{"relation":"dataset","target_id":"reported-dataset-ee26acbd6e8cf7"}],"attributes":{"origin":"independent_paper","protocol":"Interaction classifier evaluated using Boltz-1-folded input complexes; this is pipeline AUC, not DockQ.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-039","kind":"evaluation","name":"Boltz-1: Protein–ligand pose prediction","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand pose prediction"]},"source_ids":["boltz1-2025"],"links":[{"relation":"model","target_id":"reported-model-4058c43eb73b90"},{"relation":"benchmark","target_id":"reported-task-c04bb5ee6ecea6"},{"relation":"dataset","target_id":"reported-dataset-e45a5a140888ee"}],"attributes":{"origin":"author_reported","protocol":"Highest-confidence pose from five samples; precomputed MSAs up to 4,096 sequences.","version":"3 recycling rounds; 200 diffusion steps","comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-040","kind":"evaluation","name":"Ibex: Antibody loop structure prediction","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Antibody loop structure prediction"]},"source_ids":["ibex-2025"],"links":[{"relation":"model","target_id":"reported-model-2ae5fb0c147618"},{"relation":"benchmark","target_id":"reported-task-f3a12dbc0e0439"},{"relation":"dataset","target_id":"reported-dataset-b7204b005bd476"}],"attributes":{"origin":"author_reported","protocol":"Backbone RMSD after framework alignment; average over antibody test structures.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-041","kind":"evaluation","name":"Chai-1: Antibody loop structure prediction","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Antibody loop structure prediction"]},"source_ids":["ibex-2025"],"links":[{"relation":"model","target_id":"reported-model-70c732770a200f"},{"relation":"benchmark","target_id":"reported-task-f3a12dbc0e0439"},{"relation":"dataset","target_id":"reported-dataset-b7204b005bd476"}],"attributes":{"origin":"independent_paper","protocol":"Backbone RMSD after framework alignment; one seed and one diffusion trajectory.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-042","kind":"evaluation","name":"DEELIG: Protein–ligand binding affinity prediction","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand binding affinity prediction"]},"source_ids":["deelig-2021"],"links":[{"relation":"model","target_id":"reported-model-fa2da404b4d08e"},{"relation":"benchmark","target_id":"reported-task-d81be76396e644"},{"relation":"dataset","target_id":"reported-dataset-327cfcdae0c937"}],"attributes":{"origin":"author_reported","protocol":"Source paper reports DEELIG on PDBbind core set.","version":null,"comparison":{"protocol_id":null,"dataset_version":"v2016","split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-043","kind":"evaluation","name":"TOPBP (Complex): Protein–ligand binding affinity prediction","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand binding affinity prediction"]},"source_ids":["deelig-2021"],"links":[{"relation":"model","target_id":"reported-model-0068c3eff1bf7b"},{"relation":"benchmark","target_id":"reported-task-d81be76396e644"},{"relation":"dataset","target_id":"reported-dataset-327cfcdae0c937"}],"attributes":{"origin":"paper_compilation","protocol":"Source table compiles a previously published comparator; protocol equivalence is not established.","version":null,"comparison":{"protocol_id":null,"dataset_version":"v2016","split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","original_evaluation":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-044","kind":"evaluation","name":"MolAS: Physically valid protein–ligand pose selection","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Physically valid protein–ligand pose selection"]},"source_ids":["molas-2026"],"links":[{"relation":"model","target_id":"reported-model-0eb4b0535b58e3"},{"relation":"benchmark","target_id":"reported-task-d1c46526c39983"},{"relation":"dataset","target_id":"reported-dataset-4b6c13924d4256"}],"attributes":{"origin":"author_reported","protocol":"Averaged five-fold algorithm-selection performance on PoseBusters; joint RMSD and validity criterion.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-045","kind":"evaluation","name":"Single best solver: Physically valid protein–ligand pose selection","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Physically valid protein–ligand pose selection"]},"source_ids":["molas-2026"],"links":[{"relation":"model","target_id":"reported-model-028e4bb9baa074"},{"relation":"benchmark","target_id":"reported-task-d1c46526c39983"},{"relation":"dataset","target_id":"reported-dataset-4b6c13924d4256"}],"attributes":{"origin":"independent_paper","protocol":"Single best solver baseline under the same averaged five-fold selection test.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-046","kind":"evaluation","name":"AutoDock Vina holo: Intrinsically disordered protein ensemble docking","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Intrinsically disordered protein ensemble docking"]},"source_ids":["ensemble-idp-docking-2025"],"links":[{"relation":"model","target_id":"reported-model-1587ab674d30a2"},{"relation":"benchmark","target_id":"reported-task-dec9e0f5e3da2a"},{"relation":"dataset","target_id":"reported-dataset-00201f65c32f6d"}],"attributes":{"origin":"independent_paper","protocol":"Fraction of docked frames within 3 Å of MD-observed bound pose; holo protein ensemble.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-047","kind":"evaluation","name":"DiffDock holo: Intrinsically disordered protein ensemble docking","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Intrinsically disordered protein ensemble docking"]},"source_ids":["ensemble-idp-docking-2025"],"links":[{"relation":"model","target_id":"reported-model-75e4e5e5965320"},{"relation":"benchmark","target_id":"reported-task-dec9e0f5e3da2a"},{"relation":"dataset","target_id":"reported-dataset-00201f65c32f6d"}],"attributes":{"origin":"independent_paper","protocol":"Fraction of docked frames within 3 Å of MD-observed bound pose; holo protein ensemble.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-048","kind":"evaluation","name":"AK-score-ensemble: Protein–ligand binding affinity scoring","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand binding affinity scoring"]},"source_ids":["akscore-2020"],"links":[{"relation":"model","target_id":"reported-model-54be8a811c206e"},{"relation":"benchmark","target_id":"reported-task-a78312d5df6dad"},{"relation":"dataset","target_id":"reported-dataset-f18fcc23dfa798"}],"attributes":{"origin":"author_reported","protocol":"CASF-2016 scoring-power evaluation.","version":"ensemble; learning rate 0.0007","comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-049","kind":"evaluation","name":"AK-score-single: Protein–ligand binding affinity scoring","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand binding affinity scoring"]},"source_ids":["akscore-2020"],"links":[{"relation":"model","target_id":"reported-model-de89576d8d316b"},{"relation":"benchmark","target_id":"reported-task-a78312d5df6dad"},{"relation":"dataset","target_id":"reported-dataset-f18fcc23dfa798"}],"attributes":{"origin":"author_reported","protocol":"CASF-2016 scoring-power evaluation.","version":"single; learning rate 0.0007","comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-050","kind":"evaluation","name":"PMF + ECFP + PF (LightGBM): Protein–ligand binding energy prediction","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand binding energy prediction"]},"source_ids":["fingerprint-scoring-2022"],"links":[{"relation":"model","target_id":"reported-model-3c196326586fa7"},{"relation":"benchmark","target_id":"reported-task-94802534b7026d"},{"relation":"dataset","target_id":"reported-dataset-16d01b5ef88e84"}],"attributes":{"origin":"author_reported","protocol":"Binding-energy model using ligand and protein fingerprints with LightGBM.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-051","kind":"evaluation","name":"PMF (LASSO): Protein–ligand binding energy prediction","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand binding energy prediction"]},"source_ids":["fingerprint-scoring-2022"],"links":[{"relation":"model","target_id":"reported-model-8100b3de6c7811"},{"relation":"benchmark","target_id":"reported-task-94802534b7026d"},{"relation":"dataset","target_id":"reported-dataset-16d01b5ef88e84"}],"attributes":{"origin":"author_reported","protocol":"PMF-only LASSO baseline evaluated by the same authors.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-001","kind":"evaluation","name":"ARSENAL+ChromBPNet: regulatory-variant scoring","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["regulatory-variant scoring"]},"source_ids":["arsenal-regulatory-dna-2026"],"links":[{"relation":"model","target_id":"reported-model-ee1ae8162c7d67"},{"relation":"benchmark","target_id":"reported-task-b9199a30a0bcb2"},{"relation":"dataset","target_id":"reported-dataset-158b121281b650"}],"attributes":{"origin":"author_reported","protocol":"Supervised ChromBPNet variant scoring with ARSENAL motif-discovery regularization","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-002","kind":"evaluation","name":"PlantCAD2: cross-species conservation prediction","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["cross-species conservation prediction"]},"source_ids":["plantcad2-2025"],"links":[{"relation":"model","target_id":"reported-model-65059c3a806306"},{"relation":"benchmark","target_id":"reported-task-3109f8d0f2b7b5"},{"relation":"dataset","target_id":"reported-dataset-b6d0ebaca196a6"}],"attributes":{"origin":"author_reported","protocol":"Zero-shot score for conserved versus non-conserved sites from alignments of 35 Andropogoneae genomes","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-003","kind":"evaluation","name":"Stacking-Auto: enhancer prediction","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["enhancer prediction"]},"source_ids":["hi-enhancer-2025"],"links":[{"relation":"model","target_id":"reported-model-0829aff5471d4b"},{"relation":"benchmark","target_id":"reported-task-22024610c4d658"},{"relation":"dataset","target_id":"reported-dataset-a8610f2b80cdf0"}],"attributes":{"origin":"author_reported","protocol":"Two-stage Hi-Enhancer system; paper Table 2 method comparison","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-004","kind":"evaluation","name":"position-aware CNN: enhancer prediction","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["enhancer prediction"]},"source_ids":["enhancer-position-encoding-2024"],"links":[{"relation":"model","target_id":"reported-model-d2c81acf1c42c4"},{"relation":"benchmark","target_id":"reported-task-64607443a9ba15"},{"relation":"dataset","target_id":"reported-dataset-a03b8e9efde37b"}],"attributes":{"origin":"author_reported","protocol":"Nucleotide position-aware feature encoding; average assessment of CNN classifier","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-005","kind":"evaluation","name":"ADAR-GPT continual: A-to-I RNA editing site prediction","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["A-to-I RNA editing site prediction"]},"source_ids":["adar-gpt-editing-2026"],"links":[{"relation":"model","target_id":"reported-model-1d2aa9880a1c77"},{"relation":"benchmark","target_id":"reported-task-d635fc6c281a27"},{"relation":"dataset","target_id":"reported-dataset-236eaa4e55147f"}],"attributes":{"origin":"author_reported","protocol":"Curriculum plus 15% fine-tuning; 201-nt sequence windows; decision threshold 0.5","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":"15% validation set","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-006","kind":"evaluation","name":"R3Design: RNA sequence design","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA sequence design"]},"source_ids":["r3design-2025"],"links":[{"relation":"model","target_id":"reported-model-1b5fa066945d3d"},{"relation":"benchmark","target_id":"reported-task-df18c710f45213"},{"relation":"dataset","target_id":"reported-dataset-71614d99b3099f"}],"attributes":{"origin":"author_reported","protocol":"Tertiary-structure-conditioned RNA sequence design; external Rfam assessment","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":"external","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-007","kind":"evaluation","name":"CUPID Data-aug-Avg: non-coding RNA pairwise interaction prediction","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["non-coding RNA pairwise interaction prediction"]},"source_ids":["cupid-rna-interactions-2026"],"links":[{"relation":"model","target_id":"reported-model-2894d253c5e8a8"},{"relation":"benchmark","target_id":"reported-task-c7ce06b753b8b6"},{"relation":"dataset","target_id":"reported-dataset-32ccef507a1dd7"}],"attributes":{"origin":"author_reported","protocol":"Data augmentation with average pooling for molecule-level ncRNA embeddings","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-008","kind":"evaluation","name":"ProteinBERT LLM-encoding model: mRNA-protein interaction prediction","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["mRNA-protein interaction prediction"]},"source_ids":["mrna-protein-diversity-2026"],"links":[{"relation":"model","target_id":"reported-model-41ae49bb40ed8e"},{"relation":"benchmark","target_id":"reported-task-d7e6274011946e"},{"relation":"dataset","target_id":"reported-dataset-2e87449871ca47"}],"attributes":{"origin":"author_reported","protocol":"LLM encoding of protein partner; RBP-aware partition tests generalization to unseen protein diversity","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":"RBP-aware test set","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-009","kind":"evaluation","name":"ESM2 650M: human-versus-viral protein classification","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["human-versus-viral protein classification"]},"source_ids":["viral-immune-mimicry-2025"],"links":[{"relation":"model","target_id":"reported-model-4c73500c39e9d0"},{"relation":"benchmark","target_id":"reported-task-53506fe386e4a1"},{"relation":"dataset","target_id":"reported-dataset-43f24c4dfb7351"}],"attributes":{"origin":"author_reported","protocol":"ESM2 650M embedding-based human-virus classifier","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-010","kind":"evaluation","name":"ProtT5 embeddings + ensemble classifier: protein-protein binding-site prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["protein-protein binding-site prediction"]},"source_ids":["protein-binding-sites-2023"],"links":[{"relation":"model","target_id":"reported-model-fdac4c1ec8a433"},{"relation":"benchmark","target_id":"reported-task-f0ed5188dbb6d4"},{"relation":"dataset","target_id":"reported-dataset-f08b1a60aebeeb"}],"attributes":{"origin":"author_reported","protocol":"Explainable ensemble binding-site predictor using ProtT5 features","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-011","kind":"evaluation","name":"CLAPE-SMB with ESM-2: protein-small molecule binding-site prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["protein-small molecule binding-site prediction"]},"source_ids":["clape-smb-2024"],"links":[{"relation":"model","target_id":"reported-model-57dbab30462150"},{"relation":"benchmark","target_id":"reported-task-b181ed450cdd41"},{"relation":"dataset","target_id":"reported-dataset-701d910b02d25c"}],"attributes":{"origin":"author_reported","protocol":"Contrastive CLAPE-SMB binding-site predictor with ESM-2 feature extractor","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-012","kind":"evaluation","name":"Vaxign-DL + ESM: vaccine-antigen candidate prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["vaccine-antigen candidate prediction"]},"source_ids":["vaxign-esm-2024"],"links":[{"relation":"model","target_id":"reported-model-8ad3e0cefde796"},{"relation":"benchmark","target_id":"reported-task-47465954d606e6"},{"relation":"dataset","target_id":"reported-dataset-b91c871eb7740a"}],"attributes":{"origin":"author_reported","protocol":"Combined skip architecture, four layers, ESM-generated sequence features","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-013","kind":"evaluation","name":"scGPT + residual geometry: gene-regulatory signal prediction","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["gene-regulatory signal prediction"]},"source_ids":["single-cell-residual-geometry-2026"],"links":[{"relation":"model","target_id":"reported-model-d6a7fa854437e8"},{"relation":"benchmark","target_id":"reported-task-99afd88cb12895"},{"relation":"dataset","target_id":"reported-dataset-d9fdd8dc7a0184"}],"attributes":{"origin":"author_reported","protocol":"Asymmetric extraction, PCA-64 centered cosine geometry added to scGPT baseline","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-014","kind":"evaluation","name":"GREmLN: cell-type annotation","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["cell-type annotation"]},"source_ids":["gremln-2026"],"links":[{"relation":"model","target_id":"reported-model-53d6515bcc1f39"},{"relation":"benchmark","target_id":"reported-task-6312c8a7ac045e"},{"relation":"dataset","target_id":"reported-dataset-59def895fbdbb4"}],"attributes":{"origin":"author_reported","protocol":"Zero-shot cell-type annotation using pre-trained cellular graph foundation model","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":"zero-shot","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-015","kind":"evaluation","name":"Cell-DINO ViT-L: protein localization classification","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["protein localization classification"]},"source_ids":["cell-dino-2025"],"links":[{"relation":"model","target_id":"reported-model-808b23c65fbc89"},{"relation":"benchmark","target_id":"reported-task-7621fa1be55362"},{"relation":"dataset","target_id":"reported-dataset-d356eac961cb69"}],"attributes":{"origin":"author_reported","protocol":"Self-supervised microscopy embedding pre-trained on HPA-FoV; downstream protein-localization classifier. Dataset-specific pretraining; the paper does not claim a general-purpose foundation model that generalizes beyond these benchmarks.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-016","kind":"evaluation","name":"scGen: differentially expressed gene identification","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["differentially expressed gene identification"]},"source_ids":["insilico-perturbation-auprc-2025"],"links":[{"relation":"model","target_id":"reported-model-fcf2cd29a81aae"},{"relation":"benchmark","target_id":"reported-task-003d746a129c9b"},{"relation":"dataset","target_id":"reported-dataset-7bf2cf7d2b2d01"}],"attributes":{"origin":"independent_paper","protocol":"In-silico perturbation assessment with precision sampled at fixed 50% recall","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":"CD14+Mono","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-017","kind":"evaluation","name":"TCINet + HTRS: pathogen detection","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["pathogen detection"]},"source_ids":["metagenomic-pathogens-2025"],"links":[{"relation":"model","target_id":"reported-model-4ce8cae0f2eafc"},{"relation":"benchmark","target_id":"reported-task-d3fd502fdc2b38"},{"relation":"dataset","target_id":"reported-dataset-1c4c71078ffe01"}],"attributes":{"origin":"author_reported","protocol":"Taxonomy-constrained inference network with hierarchical taxonomy representation","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-018","kind":"evaluation","name":"DETIRE: viral sequence detection","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["viral sequence detection"]},"source_ids":["detire-viral-metagenomes-2023"],"links":[{"relation":"model","target_id":"reported-model-6d9dbac97852d8"},{"relation":"benchmark","target_id":"reported-task-3d4dec23120fef"},{"relation":"dataset","target_id":"reported-dataset-0b54f42a987b1d"}],"attributes":{"origin":"author_reported","protocol":"Hybrid deep learning virus-fragment classifier on paper testing dataset","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":"test","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-019","kind":"evaluation","name":"PC-mer + LR: metagenomic genus classification","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["metagenomic genus classification"]},"source_ids":["pc-mer-2024"],"links":[{"relation":"model","target_id":"reported-model-688eb780ef7d2e"},{"relation":"benchmark","target_id":"reported-task-4420dcdfe8338d"},{"relation":"dataset","target_id":"reported-dataset-d28955d5872903"}],"attributes":{"origin":"author_reported","protocol":"k=8 PC-mer feature extraction with logistic regression on AMP genus-classification dataset","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":"genus-level","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-020","kind":"evaluation","name":"MDL4Microbiome: microbiome disease-state classification","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["microbiome disease-state classification"]},"source_ids":["mdl4microbiome-2022"],"links":[{"relation":"model","target_id":"reported-model-e6ba198c2ac996"},{"relation":"benchmark","target_id":"reported-task-e2009c35eabd69"},{"relation":"dataset","target_id":"reported-dataset-bd3f98e2eeb5d3"}],"attributes":{"origin":"author_reported","protocol":"Multimodal deep learning model on colorectal-cancer versus healthy microbiome samples","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-021","kind":"evaluation","name":"binding-affinity meta-model: protein-ligand binding affinity prediction","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["protein-ligand binding affinity prediction"]},"source_ids":["ligand-affinity-meta-model-2024"],"links":[{"relation":"model","target_id":"reported-model-df0efcc0346224"},{"relation":"benchmark","target_id":"reported-task-77a32496ce8fe6"},{"relation":"dataset","target_id":"reported-dataset-17132fbabd7683"}],"attributes":{"origin":"author_reported","protocol":"Sequence-or-structure meta-model; predicts ln(Kd/Ki) using docked and deep-learning components","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":"core benchmark","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-022","kind":"evaluation","name":"DeepInterAware: antigen-antibody HIV neutralization prediction","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["antigen-antibody HIV neutralization prediction"]},"source_ids":["deepinteraware-2025"],"links":[{"relation":"model","target_id":"reported-model-8d2c291733dfe1"},{"relation":"benchmark","target_id":"reported-task-45ead9a1eddf8d"},{"relation":"dataset","target_id":"reported-dataset-50f0bdb7cf9ca4"}],"attributes":{"origin":"author_reported","protocol":"Sequence-based interface-aware model, antibody-unseen split","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":"antibody-unseen","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-023","kind":"evaluation","name":"TransBind: transcription-factor DNA binding-site prediction","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["transcription-factor DNA binding-site prediction"]},"source_ids":["transbind-2026"],"links":[{"relation":"model","target_id":"reported-model-81b0394d5ac3e8"},{"relation":"benchmark","target_id":"reported-task-ac191e878dff5e"},{"relation":"dataset","target_id":"reported-dataset-034c60a2dabc73"}],"attributes":{"origin":"author_reported","protocol":"Integrates protein and DNA embeddings for TFBS prediction on paper test dataset","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":"test","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-024","kind":"evaluation","name":"ESM2_AMPS: protein-protein interaction prediction","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["protein-protein interaction prediction"]},"source_ids":["esm2-amp-2025"],"links":[{"relation":"model","target_id":"reported-model-67ea6bd77b2ed1"},{"relation":"benchmark","target_id":"reported-task-09c3100b77dcc5"},{"relation":"dataset","target_id":"reported-dataset-9135087a16af1c"}],"attributes":{"origin":"author_reported","protocol":"ESM2-derived embeddings plus paper interaction predictor","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"fingerprint-scoring-2022","kind":"source","name":"Machine-Learning- and Knowledge-Based Scoring Functions Incorporating Ligand and Protein Fingerprints","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC9178954/","version":"PMC archival version PMC9178954.1","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.1021/acsomega.2c02822","publication_status":"peer_reviewed","year":2022,"artifact_sha256":"47bd60c6392b801095fdb604de06c0d4bda6f555bae58e9955e491a5abf60576","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC9178954/fullTextXML","artifact_retrieved_at":"2026-09-16T10:44:03.439460+00:00","legacy_paper":{"id":"fingerprint-scoring-2022","title":"Machine-Learning- and Knowledge-Based Scoring Functions Incorporating Ligand and Protein Fingerprints","year":2022,"publication_status":"peer_reviewed","version":"PMC archival version PMC9178954.1","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC9178954/","primary_domain":"molecular-interactions","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: ACS Omega; PMC ID: PMC9178954.","doi":"10.1021/acsomega.2c02822"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"fujisan-2024","kind":"source","name":"Enhanced prediction of protein functional identity through the integration of sequence and structural features","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["proteins-complexes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11609699/","version":"PMC11609699.1","retrieved_at":"2026-09-15T23:25:00Z","doi":"10.1016/j.csbj.2024.11.028","publication_status":"peer_reviewed","year":2024,"artifact_sha256":"db33e0542005ffae00cd644dfe697185b94c8823d5aee2768620a6db0c48e56f","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11609699/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:35.728Z","legacy_paper":{"id":"fujisan-2024","title":"Enhanced prediction of protein functional identity through the integration of sequence and structural features","year":2024,"publication_status":"peer_reviewed","version":"PMC11609699.1","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11609699/","primary_domain":"proteins-complexes","retrieved_utc":"2026-09-15T23:25:00Z","notes":"Primary full text via Europe PMC XML; venue: Computational and Structural Biotechnology Journal; PMC ID: PMC11609699.","doi":"10.1016/j.csbj.2024.11.028"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"fusion-breakpoint-foundation-models-2026","kind":"source","name":"Benchmarking genomic foundation models for binary classification of gene fusion breakpoints from DNA sequences","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["dna-genomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13182013/","version":"journal full text in PMC","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.1186/s13040-026-00553-1","publication_status":"peer_reviewed","year":2026,"artifact_sha256":"0f4d9de77f1e39cfd2164a20653d86370767da684dc22d17e09f589761abeb5f","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13182013/fullTextXML","artifact_retrieved_at":"2026-09-16T10:38:57.558209+00:00","legacy_paper":{"id":"fusion-breakpoint-foundation-models-2026","title":"Benchmarking genomic foundation models for binary classification of gene fusion breakpoints from DNA sequences","year":2026,"publication_status":"peer_reviewed","version":"journal full text in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13182013/","primary_domain":"dna-genomes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.1186/s13040-026-00553-1","notes":"Numeric result checked against Table 2 in primary full-text XML; journal/source: BioData Mining."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"genept-2024","kind":"source","name":"GenePT: A Simple But Effective Foundation Model for Genes and Cells Built From ChatGPT","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["cells-tissues"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC10614824/","version":"PMC archival version PMC10614824.2","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.1101/2023.10.16.562533","publication_status":"preprint","year":2024,"artifact_sha256":"230a2ec55458d9243eaeeebf3244df7409eb02d47f4b809ee56a06dcb6fdd047","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC10614824/fullTextXML","artifact_retrieved_at":"2026-09-16T10:44:03.399274+00:00","legacy_paper":{"id":"genept-2024","title":"GenePT: A Simple But Effective Foundation Model for Genes and Cells Built From ChatGPT","year":2024,"publication_status":"preprint","version":"PMC archival version PMC10614824.2","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC10614824/","primary_domain":"cells-tissues","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: bioRxiv; PMC ID: PMC10614824.","doi":"10.1101/2023.10.16.562533"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"genomeocean-2025","kind":"source","name":"GenomeOcean: An Efficient Genome Foundation Model Trained on Large-Scale Metagenomic Assemblies","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11838515/","version":"preprint archived 2025-02-05","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.1101/2025.01.30.635558","publication_status":"preprint","year":2025,"artifact_sha256":"3cc0df52522fccda23e3958f069c916b87ee50bb5c9a992fa37e25256546e145","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11838515/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:55.224Z","legacy_paper":{"id":"genomeocean-2025","title":"GenomeOcean: An Efficient Genome Foundation Model Trained on Large-Scale Metagenomic Assemblies","year":2025,"publication_status":"preprint","version":"preprint archived 2025-02-05","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11838515/","primary_domain":"microbes-communities","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: bioRxiv; PMC ID: PMC11838515.","doi":"10.1101/2025.01.30.635558"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"genomic-tokenizer-selection-2025","kind":"source","name":"The impact of tokenizer selection in genomic language models","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["dna-genomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12453675/","version":"journal full text in PMC","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.1093/bioinformatics/btaf456","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"0a01c36fdd63f3f6db509777e61c3f87e8a298c810f8aef7974915aaa0655342","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12453675/fullTextXML","artifact_retrieved_at":"2026-09-16T10:38:57.558210+00:00","legacy_paper":{"id":"genomic-tokenizer-selection-2025","title":"The impact of tokenizer selection in genomic language models","year":2025,"publication_status":"peer_reviewed","version":"journal full text in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12453675/","primary_domain":"dna-genomes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.1093/bioinformatics/btaf456","notes":"Final Bioinformatics journal article Table 2, Caduceus (char) Regulatory MCC 0.778 checked directly; same study also has a bioRxiv manuscript."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"gremln-2026","kind":"source","name":"GREmLN: A Cellular Graph Structure Aware Transcriptomics Foundation Model","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["cells-tissues"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13060794/","version":"preprint version in PMC","retrieved_at":"2026-09-15T23:37:05Z","doi":"10.1101/2025.07.03.663009","publication_status":"preprint","year":2026,"artifact_sha256":"3a20c4ededb749fc3f1120baf16dcfebe3fcb30418a91c445cfd91a7b5fdf553","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13060794/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:57.502Z","legacy_paper":{"id":"gremln-2026","title":"GREmLN: A Cellular Graph Structure Aware Transcriptomics Foundation Model","year":2026,"publication_status":"preprint","version":"preprint version in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13060794/","primary_domain":"cells-tissues","retrieved_utc":"2026-09-15T23:37:05Z","notes":"Primary full text verified via Europe PMC fullTextXML; venue: bioRxiv; PMC ID: PMC13060794. Preprint; table labels metric F1; paper does not specify macro in this row.","doi":"10.1101/2025.07.03.663009"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"gsmformer-ppi-2026","kind":"source","name":"Multimodal graph, surface, and language-based model for protein protein interaction prediction","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["proteins-complexes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12873117/","version":"journal full text in PMC","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.1038/s41598-025-34758-x","publication_status":"peer_reviewed","year":2026,"artifact_sha256":"9b364b5d73d16f2787f93f78f17dbe98b954ab9c2c64c1df960eec2e615eb3b4","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12873117/fullTextXML","artifact_retrieved_at":"2026-09-16T10:38:57.558212+00:00","legacy_paper":{"id":"gsmformer-ppi-2026","title":"Multimodal graph, surface, and language-based model for protein protein interaction prediction","year":2026,"publication_status":"peer_reviewed","version":"journal full text in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12873117/","primary_domain":"proteins-complexes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.1038/s41598-025-34758-x","notes":"Numeric result checked against Table 6 in primary full-text XML; journal/source: Scientific Reports."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"hi-enhancer-2025","kind":"source","name":"Hi-Enhancer: a two-stage framework for prediction and localization of enhancers based on Blending-KAN and Stacking-Auto models","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["dna-genomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12758598/","version":"version of record","retrieved_at":"2026-09-15T23:37:05Z","doi":"10.1093/bioinformatics/btaf441","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"c86488c9f60329b7a3c4370598e7a0a9e4c8c45d1758b87007bfc8242376b009","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12758598/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:06Z","legacy_paper":{"id":"hi-enhancer-2025","title":"Hi-Enhancer: a two-stage framework for prediction and localization of enhancers based on Blending-KAN and Stacking-Auto models","year":2025,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12758598/","primary_domain":"dna-genomes","retrieved_utc":"2026-09-15T23:37:05Z","notes":"Primary full text verified via Europe PMC fullTextXML; venue: Bioinformatics; PMC ID: PMC12758598. Task-specific enhancer predictor; not a DNA foundation model. Comparison values from older papers excluded.","doi":"10.1093/bioinformatics/btaf441"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"ibex-2025","kind":"source","name":"Conformation-aware structure prediction of antigen-recognizing immune proteins","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12710905/","version":"PMC archival version PMC12710905.1","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.1080/19420862.2025.2602217","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"caa1109bd5fe7f6be703aa9d4afd6f4f1522bcbce6b7361650eb59618c2a9e14","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12710905/fullTextXML","artifact_retrieved_at":"2026-09-16T10:44:03.426811+00:00","legacy_paper":{"id":"ibex-2025","title":"Conformation-aware structure prediction of antigen-recognizing immune proteins","year":2025,"publication_status":"peer_reviewed","version":"PMC archival version PMC12710905.1","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12710905/","primary_domain":"molecular-interactions","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: mAbs; PMC ID: PMC12710905.","doi":"10.1080/19420862.2025.2602217"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"icctax-2025","kind":"source","name":"ICCTax: a hierarchical taxonomic classifier for metagenomic sequences on a large language model","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12619997/","version":"version of record","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.1093/bioadv/vbaf257","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"2ce0b48f1cde3aea7e561d92f4d7dc1525af7439ccd16f80bec0773e8812c8ec","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12619997/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:55.373Z","legacy_paper":{"id":"icctax-2025","title":"ICCTax: a hierarchical taxonomic classifier for metagenomic sequences on a large language model","year":2025,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12619997/","primary_domain":"microbes-communities","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: Bioinformatics Advances; PMC ID: PMC12619997.","doi":"10.1093/bioadv/vbaf257"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"insilico-perturbation-auprc-2025","kind":"source","name":"AUPRC: a metric for evaluating the performance of in-silico perturbation methods in identifying differentially expressed genes","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["cells-tissues"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12400816/","version":"PMC archival version PMC12400816.1","retrieved_at":"2026-09-15T23:37:05Z","doi":"10.1093/bib/bbaf426","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"2715709d94f84744afa32cafdcaa72efd206d63af8c60afe7619b2cb90108b6b","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12400816/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:06Z","legacy_paper":{"id":"insilico-perturbation-auprc-2025","title":"AUPRC: a metric for evaluating the performance of in-silico perturbation methods in identifying differentially expressed genes","year":2025,"publication_status":"peer_reviewed","version":"PMC archival version PMC12400816.1","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12400816/","primary_domain":"cells-tissues","retrieved_utc":"2026-09-15T23:37:05Z","notes":"Primary full text verified via Europe PMC fullTextXML; venue: Briefings in Bioinformatics; PMC ID: PMC12400816. Paper benchmarks metrics and scGen perturbation method; no foundation-model result in this row.","doi":"10.1093/bib/bbaf426"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"ipromp-2025","kind":"source","name":"iPro-MP: a BERT-based model to predict multiple prokaryotic promoters","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12516880/","version":"version of record","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.1186/s13059-025-03819-9","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"d21541ee1f7a168da8e4a7c0f0e133c970cbe7bc41118f43a929f08b2fd2afd1","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12516880/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:55.361Z","legacy_paper":{"id":"ipromp-2025","title":"iPro-MP: a BERT-based model to predict multiple prokaryotic promoters","year":2025,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12516880/","primary_domain":"microbes-communities","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: Genome Biology; PMC ID: PMC12516880.","doi":"10.1186/s13059-025-03819-9"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"kmetashot-2025","kind":"source","name":"kMetaShot: a fast and reliable taxonomy classifier for metagenome-assembled genomes","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11695915/","version":"PMC archival version PMC11695915.1","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.1093/bib/bbae680","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"4584e93ea035c1170b8756a0a52cbe99fe72e70bd09b5f1dee639ee104f78247","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11695915/fullTextXML","artifact_retrieved_at":"2026-09-16T10:44:03.417367+00:00","legacy_paper":{"id":"kmetashot-2025","title":"kMetaShot: a fast and reliable taxonomy classifier for metagenome-assembled genomes","year":2025,"publication_status":"peer_reviewed","version":"PMC archival version PMC11695915.1","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11695915/","primary_domain":"microbes-communities","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: Briefings in Bioinformatics; PMC ID: PMC11695915.","doi":"10.1093/bib/bbae680"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"lambda-prophage-2026","kind":"source","name":"LAMBDA: A Prophage Detection Benchmark for Genomic Language Models","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13041943/","version":"PMC13041943.1","retrieved_at":"2026-09-15T23:25:00Z","doi":"10.64898/2026.03.26.714501","publication_status":"preprint","year":2026,"artifact_sha256":"22c2e218e87dce757907f6086a0e2ad37c13f785b34fff5bea7cfa1a6c276b16","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13041943/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:36.240Z","legacy_paper":{"id":"lambda-prophage-2026","title":"LAMBDA: A Prophage Detection Benchmark for Genomic Language Models","year":2026,"publication_status":"preprint","version":"PMC13041943.1","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13041943/","primary_domain":"microbes-communities","retrieved_utc":"2026-09-15T23:25:00Z","notes":"Primary full text via Europe PMC XML; venue: bioRxiv; PMC ID: PMC13041943.","doi":"10.64898/2026.03.26.714501"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"lazypipe-2020","kind":"source","name":"Novel NGS pipeline for virus discovery from a wide spectrum of hosts and sample types","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC7772471/","version":"version of record","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.1093/ve/veaa091","publication_status":"peer_reviewed","year":2020,"artifact_sha256":"77842d8e4f6b419e331ab5a01fdf8f9eb8604f259425d79602be896aad3d0ad1","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC7772471/fullTextXML","artifact_retrieved_at":"2026-09-16T10:44:03.412605+00:00","legacy_paper":{"id":"lazypipe-2020","title":"Novel NGS pipeline for virus discovery from a wide spectrum of hosts and sample types","year":2020,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC7772471/","primary_domain":"microbes-communities","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: Virus Evolution; PMC ID: PMC7772471.","doi":"10.1093/ve/veaa091"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"lemur-magnet-2024","kind":"source","name":"Lightweight taxonomic profiling of long-read metagenomic datasets with Lemur and Magnet","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11185576/","version":"PMC archival version PMC11185576.2","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.1101/2024.06.01.596961","publication_status":"preprint","year":2024,"artifact_sha256":"4afb9195da447916eb6f733816e3640741c7ade08ea8920d205c3be7b3cce27a","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11185576/fullTextXML","artifact_retrieved_at":"2026-09-16T10:44:03.420493+00:00","legacy_paper":{"id":"lemur-magnet-2024","title":"Lightweight taxonomic profiling of long-read metagenomic datasets with Lemur and Magnet","year":2024,"publication_status":"preprint","version":"PMC archival version PMC11185576.2","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11185576/","primary_domain":"microbes-communities","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: bioRxiv; PMC ID: PMC11185576.","doi":"10.1101/2024.06.01.596961"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"ligand-affinity-meta-model-2024","kind":"source","name":"Improved Prediction of Ligand–Protein Binding Affinities by Meta-modeling","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11632770/","version":"PMC archival version PMC11632770.1","retrieved_at":"2026-09-15T23:37:05Z","doi":"10.1021/acs.jcim.4c01116","publication_status":"peer_reviewed","year":2024,"artifact_sha256":"0be25fe75bc0b2eb8065136555763bbae5964ea3a8fdb8c5de79ff96445f6a29","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11632770/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:06Z","legacy_paper":{"id":"ligand-affinity-meta-model-2024","title":"Improved Prediction of Ligand–Protein Binding Affinities by Meta-modeling","year":2024,"publication_status":"peer_reviewed","version":"PMC archival version PMC11632770.1","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11632770/","primary_domain":"molecular-interactions","retrieved_utc":"2026-09-15T23:37:05Z","notes":"Primary full text verified via Europe PMC fullTextXML; venue: Journal of Chemical Information and Modeling; PMC ID: PMC11632770. Mixed prediction units across Table 4 comparators; only meta-model PCC recorded.","doi":"10.1021/acs.jcim.4c01116"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"lipp-2026","kind":"source","name":"The LiPP Benchmark Set for Modeling Lipid–Protein Complexes: Comparison of Co-Folding and Docking Methods","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13292216/","version":"PMC13292216.1","retrieved_at":"2026-09-15T23:25:00Z","doi":"10.1021/acs.jcim.6c01457","publication_status":"peer_reviewed","year":2026,"artifact_sha256":"6ff34f2f709a14858a3753abf9f8f6efa1e7e3c351f15c70cf64264193a9414e","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13292216/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:16.548973+00:00","legacy_paper":{"id":"lipp-2026","title":"The LiPP Benchmark Set for Modeling Lipid–Protein Complexes: Comparison of Co-Folding and Docking Methods","year":2026,"publication_status":"peer_reviewed","version":"PMC13292216.1","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13292216/","primary_domain":"molecular-interactions","retrieved_utc":"2026-09-15T23:25:00Z","notes":"Primary full text via Europe PMC XML; venue: Journal of Chemical Information and Modeling; PMC ID: PMC13292216.","doi":"10.1021/acs.jcim.6c01457"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"lit-001","kind":"result","name":"Caduceus-Ph · AUC · Human 5mC","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["Human 5mC detection"]},"source_ids":["dna-foundation-models-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-001"}],"attributes":{"printed_value":"0.783","numeric_value":"0.783","metric":"AUC","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 3, Human 5mC row, Caduceus-Ph column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:35.327Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 3, Human 5mC row, Caduceus-Ph column; cell: 0.783","artifact_sha256":"5d8ca9bcf88cc1b38ad667906a2e4699b1aefa6d31c6f49259784930353f3202","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12663285/fullTextXML"},"legacy_id":"lit-001","legacy_row":{"id":"lit-001","paper_id":"dna-foundation-models-2025","domain_id":"dna-genomes","task":"Human 5mC detection","model":"Caduceus-Ph","model_version":"","dataset":"Human 5mC","dataset_version":"","split":"","metric":"AUC","value":"0.783","unit":"unitless","uncertainty":"","protocol":"Binary epigenetic-modification classification as reported in the paper.","source_locator":"Table 3, Human 5mC row, Caduceus-Ph column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12663285/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-002","kind":"result","name":"NT-v2 · AUC · Human 5mC","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["Human 5mC detection"]},"source_ids":["dna-foundation-models-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-002"}],"attributes":{"printed_value":"0.7377","numeric_value":"0.7377","metric":"AUC","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 3, Human 5mC row, NT-v2 column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:35.327Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 3, Human 5mC row, NT-v2 column; cell: 0.7377","artifact_sha256":"5d8ca9bcf88cc1b38ad667906a2e4699b1aefa6d31c6f49259784930353f3202","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12663285/fullTextXML"},"legacy_id":"lit-002","legacy_row":{"id":"lit-002","paper_id":"dna-foundation-models-2025","domain_id":"dna-genomes","task":"Human 5mC detection","model":"NT-v2","model_version":"","dataset":"Human 5mC","dataset_version":"","split":"","metric":"AUC","value":"0.7377","unit":"unitless","uncertainty":"","protocol":"Binary epigenetic-modification classification as reported in the paper.","source_locator":"Table 3, Human 5mC row, NT-v2 column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12663285/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-003","kind":"result","name":"ENBED · Accuracy · Genomic Benchmarks Mouse Enhancers","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["Enhancer classification"]},"source_ids":["enbed-2024"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-003"}],"attributes":{"printed_value":"90.3","numeric_value":"90.3","metric":"Accuracy","metric_direction":"unknown","unit":"%","uncertainty":null,"source_locator":"Table 2, Mouse Enhancers row, ENBED column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:35.378Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 2, Mouse Enhancers row, ENBED column; cell: 90.3","artifact_sha256":"e95d4be70d32e61af5a92eda8ea66f25a2cc629e3f83e7b5241b13cde8bdb83b","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11341122/fullTextXML"},"legacy_id":"lit-003","legacy_row":{"id":"lit-003","paper_id":"enbed-2024","domain_id":"dna-genomes","task":"Enhancer classification","model":"ENBED","model_version":"","dataset":"Genomic Benchmarks Mouse Enhancers","dataset_version":"","split":"","metric":"Accuracy","value":"90.3","unit":"%","uncertainty":"","protocol":"Reported Genomic Benchmarks classification accuracy.","source_locator":"Table 2, Mouse Enhancers row, ENBED column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11341122/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-004","kind":"result","name":"ENBED (GRCh38) · Accuracy · Genomic Benchmarks Mouse Enhancers","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["Enhancer classification"]},"source_ids":["enbed-2024"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-004"}],"attributes":{"printed_value":"81.1","numeric_value":"81.1","metric":"Accuracy","metric_direction":"unknown","unit":"%","uncertainty":null,"source_locator":"Table 2, Mouse Enhancers row, ENBED (GRCh38) column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:35.378Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 2, Mouse Enhancers row, ENBED (GRCh38) column; cell: 81.1","artifact_sha256":"e95d4be70d32e61af5a92eda8ea66f25a2cc629e3f83e7b5241b13cde8bdb83b","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11341122/fullTextXML"},"legacy_id":"lit-004","legacy_row":{"id":"lit-004","paper_id":"enbed-2024","domain_id":"dna-genomes","task":"Enhancer classification","model":"ENBED (GRCh38)","model_version":"","dataset":"Genomic Benchmarks Mouse Enhancers","dataset_version":"","split":"","metric":"Accuracy","value":"81.1","unit":"%","uncertainty":"","protocol":"ENBED trained on GRCh38; reported Genomic Benchmarks classification accuracy.","source_locator":"Table 2, Mouse Enhancers row, ENBED (GRCh38) column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11341122/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-005","kind":"result","name":"DNABERT-2 · Accuracy · KEx","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["G-quadruplex classification"]},"source_ids":["quadruplex-llm-benchmark-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-005"}],"attributes":{"printed_value":"97.0","numeric_value":"97.0","metric":"Accuracy","metric_direction":"unknown","unit":"%","uncertainty":"± 0.5","source_locator":"Table 5, DNABERT-2 (117 M) row, Accuracy column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:35.379Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 5, DNABERT-2 (117 M) row, Accuracy column; cell: 97.0 ± 0.5","artifact_sha256":"c3d7c6d068d3c11a9c8255a932197ece3804d94e8b2d4f858bea373a1b6eb32f","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11953744/fullTextXML"},"legacy_id":"lit-005","legacy_row":{"id":"lit-005","paper_id":"quadruplex-llm-benchmark-2025","domain_id":"dna-genomes","task":"G-quadruplex classification","model":"DNABERT-2","model_version":"117M","dataset":"KEx","dataset_version":"","split":"","metric":"Accuracy","value":"97.0","unit":"%","uncertainty":"± 0.5","protocol":"Pretrained model evaluated on KEx as reported in Table 5.","source_locator":"Table 5, DNABERT-2 (117 M) row, Accuracy column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11953744/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"lit-006","kind":"result","name":"Caduceus · Accuracy · KEx","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["G-quadruplex classification"]},"source_ids":["quadruplex-llm-benchmark-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-006"}],"attributes":{"printed_value":"95.0","numeric_value":"95.0","metric":"Accuracy","metric_direction":"unknown","unit":"%","uncertainty":"± 0.5","source_locator":"Table 5, Caduceus (8 M) row, Accuracy column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:35.379Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 5, Caduceus (8 M) row, Accuracy column; cell: 95.0 ± 0.5","artifact_sha256":"c3d7c6d068d3c11a9c8255a932197ece3804d94e8b2d4f858bea373a1b6eb32f","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11953744/fullTextXML"},"legacy_id":"lit-006","legacy_row":{"id":"lit-006","paper_id":"quadruplex-llm-benchmark-2025","domain_id":"dna-genomes","task":"G-quadruplex classification","model":"Caduceus","model_version":"8M","dataset":"KEx","dataset_version":"","split":"","metric":"Accuracy","value":"95.0","unit":"%","uncertainty":"± 0.5","protocol":"Pretrained model evaluated on KEx as reported in Table 5.","source_locator":"Table 5, Caduceus (8 M) row, Accuracy column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11953744/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"lit-007","kind":"result","name":"HyenaDNA · AUROC · DNALongBench ETGP","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["Enhancer-target gene prediction"]},"source_ids":["dnalongbench-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-007"}],"attributes":{"printed_value":"0.828","numeric_value":"0.828","metric":"AUROC","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 3, HyenaDNA row, ETGP column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.492545+00:00","notes":"ETGP is the first numeric column, separate from the six CMP cell-type columns and average; caption defines ETGP AUROC. This verifies the central score at its source location, not every metadata field or an experimental reproduction.","evidence":"{\"table_xml_id\": \"T3\", \"row_cells\": [\"HyenaDNA\", \"0.828\", \"0.139\", \"0.122\", \"0.099\", \"0.097\", \"0.118\", \"0.115\"], \"selected_cell_zero_based\": 1, \"selected_cell_xml\": \"