{"id":"2ome-lm-2025","kind":"source","name":"2OMe-LM: predicting 2′-O-methylation sites in human RNA using a pre-trained RNA language model","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["rna-transcriptomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12342186/","version":"journal full text in PMC","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.1093/bioinformatics/btaf417","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"54fe4db6f35c03d0d4f3ef4da720eb26a832199372c56d0956609ff07af750ee","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12342186/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:38.332Z","legacy_paper":{"id":"2ome-lm-2025","title":"2OMe-LM: predicting 2′-O-methylation sites in human RNA using a pre-trained RNA language model","year":2025,"publication_status":"peer_reviewed","version":"journal full text in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12342186/","primary_domain":"rna-transcriptomes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.1093/bioinformatics/btaf417","notes":"Numeric result checked against Table 1. in primary full-text XML; journal/source: Bioinformatics."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"adar-gpt-editing-2026","kind":"source","name":"ADAR-GPT: A continually fine-tuned language model for predicting A-to-I RNA editing sites","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["rna-transcriptomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12798952/","version":"version of record","retrieved_at":"2026-09-15T23:37:05Z","doi":"10.1073/pnas.2529073123","publication_status":"peer_reviewed","year":2026,"artifact_sha256":"cc8c7eb928f246f1f347a8822f614cd3475381c35eef6d579032ce441580198e","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12798952/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:06Z","legacy_paper":{"id":"adar-gpt-editing-2026","title":"ADAR-GPT: A continually fine-tuned language model for predicting A-to-I RNA editing sites","year":2026,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12798952/","primary_domain":"rna-transcriptomes","retrieved_utc":"2026-09-15T23:37:05Z","notes":"Primary full text verified via Europe PMC fullTextXML; venue: Proceedings of the National Academy of Sciences of the United States of America; PMC ID: PMC12798952. RNA editing site benchmark on a restricted liver validation set.","doi":"10.1073/pnas.2529073123"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"akscore-2020","kind":"source","name":"AK-Score: Accurate Protein-Ligand Binding Affinity Prediction Using an Ensemble of 3D-Convolutional Neural Networks","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC7697539/","version":"version of record","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.3390/ijms21228424","publication_status":"peer_reviewed","year":2020,"artifact_sha256":"40cfd28dcd587599768ec99a6590ec593486475ff01c7b1d1f229b44aa91bf8d","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC7697539/fullTextXML","artifact_retrieved_at":"2026-09-16T10:44:03.436853+00:00","legacy_paper":{"id":"akscore-2020","title":"AK-Score: Accurate Protein-Ligand Binding Affinity Prediction Using an Ensemble of 3D-Convolutional Neural Networks","year":2020,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC7697539/","primary_domain":"molecular-interactions","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: International Journal of Molecular Sciences; PMC ID: PMC7697539.","doi":"10.3390/ijms21228424"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"antibody-deamidation-plm-2024","kind":"source","name":"The Accurate Prediction of Antibody Deamidations by Combining High-Throughput Automated Peptide Mapping and Protein Language Model-Based Deep Learning","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["proteins-complexes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11417914/","version":"journal full text in PMC","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.3390/antib13030074","publication_status":"peer_reviewed","year":2024,"artifact_sha256":"aa049f6d78e29540ba902a0d3b7f53d49e9833e4dad9869f79ac27664e8c150b","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11417914/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:38.478Z","legacy_paper":{"id":"antibody-deamidation-plm-2024","title":"The Accurate Prediction of Antibody Deamidations by Combining High-Throughput Automated Peptide Mapping and Protein Language Model-Based Deep Learning","year":2024,"publication_status":"peer_reviewed","version":"journal full text in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11417914/","primary_domain":"proteins-complexes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.3390/antib13030074","notes":"Numeric result checked against Table 1 in primary full-text XML; journal/source: Antibodies."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"antibody-flexibility-2025","kind":"source","name":"Enhancing antibody-antigen interaction prediction with atomic flexibility","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12530544/","version":"version of record","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.1371/journal.pcbi.1013576","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"57e64694c69052ed0495570e12ebfb4bb6c0ad152219f23827cd4b1cb53450ef","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12530544/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:55.400Z","legacy_paper":{"id":"antibody-flexibility-2025","title":"Enhancing antibody-antigen interaction prediction with atomic flexibility","year":2025,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12530544/","primary_domain":"molecular-interactions","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: PLOS Computational Biology; PMC ID: PMC12530544.","doi":"10.1371/journal.pcbi.1013576"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"arsenal-regulatory-dna-2026","kind":"source","name":"Short-Context Regulatory DNA Language Models with Motif-Discovery Regularization","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["dna-genomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12889687/","version":"preprint version in PMC","retrieved_at":"2026-09-15T23:37:05Z","doi":"10.64898/2026.02.05.703637","publication_status":"preprint","year":2026,"artifact_sha256":"4a264956e47fc633aaff6573aac368dc691dd5de709b27c7421c078608ff542a","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12889687/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:06Z","legacy_paper":{"id":"arsenal-regulatory-dna-2026","title":"Short-Context Regulatory DNA Language Models with Motif-Discovery Regularization","year":2026,"publication_status":"preprint","version":"preprint version in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12889687/","primary_domain":"dna-genomes","retrieved_utc":"2026-09-15T23:37:05Z","notes":"Primary full text verified via Europe PMC fullTextXML; venue: bioRxiv; PMC ID: PMC12889687. Preprint; result is a supervised downstream model rather than a general-purpose DNA foundation model.","doi":"10.64898/2026.02.05.703637"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"b2-2ome-lm-2025","kind":"result","name":"2OMe-LM · AUC · human RNA 2OMe sites","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["human RNA 2-prime-O-methylation site prediction"]},"source_ids":["2ome-lm-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-2ome-lm-2025"}],"attributes":{"printed_value":"0.919","numeric_value":"0.919","metric":"AUC","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 1, 2OMe-LM row, AUC column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:38.332Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 1, 2OMe-LM row, AUC column; cell: 0.919","artifact_sha256":"54fe4db6f35c03d0d4f3ef4da720eb26a832199372c56d0956609ff07af750ee","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12342186/fullTextXML"},"legacy_id":"b2-2ome-lm-2025","legacy_row":{"id":"b2-2ome-lm-2025","paper_id":"2ome-lm-2025","domain_id":"rna-transcriptomes","task":"human RNA 2-prime-O-methylation site prediction","model":"2OMe-LM","model_version":"not stated in table","dataset":"human RNA 2OMe sites","dataset_version":"","split":"5-fold cross-validation","metric":"AUC","value":"0.919","unit":"fraction","uncertainty":"","protocol":"pretrained RNA language model predictor","source_locator":"Table 1, 2OMe-LM row, AUC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12342186/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-antibody-deamidation-plm-2024","kind":"result","name":"ESM-2 650M embeddings + classifier · accuracy · antibody peptide-mapping training dataset","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["antibody deamidation-site prediction"]},"source_ids":["antibody-deamidation-plm-2024"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-antibody-deamidation-plm-2024"}],"attributes":{"printed_value":"0.944","numeric_value":"0.944","metric":"accuracy","metric_direction":"unknown","unit":"fraction","uncertainty":"± 0.012","source_locator":"Table 1, Global embeddings only row, Accuracy column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:38.478Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 1, Global embeddings only row, Accuracy column; cell: 0.944 ± 0.012","artifact_sha256":"aa049f6d78e29540ba902a0d3b7f53d49e9833e4dad9869f79ac27664e8c150b","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11417914/fullTextXML"},"legacy_id":"b2-antibody-deamidation-plm-2024","legacy_row":{"id":"b2-antibody-deamidation-plm-2024","paper_id":"antibody-deamidation-plm-2024","domain_id":"proteins-complexes","task":"antibody deamidation-site prediction","model":"ESM-2 650M embeddings + classifier","model_version":"esm2_t33_650m_UR50D","dataset":"antibody peptide-mapping training dataset","dataset_version":"","split":"fivefold stratified CV","metric":"accuracy","value":"0.944","unit":"fraction","uncertainty":"± 0.012","protocol":"global contextual embeddings only","source_locator":"Table 1, Global embeddings only row, Accuracy column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11417914/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"b2-barcodebert-2026","kind":"result","name":"BarcodeBERT (4–4-4) · accuracy · DNA barcodes of unseen species","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["unseen-species genus classification"]},"source_ids":["barcodebert-2026"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-barcodebert-2026"}],"attributes":{"printed_value":"78.5","numeric_value":"78.5","metric":"accuracy","metric_direction":"unknown","unit":"percent","uncertainty":null,"source_locator":"Table 1, BarcodeBERT (4–4-4) row, unseen-species genus-level 1-NN Acc (%) column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558051+00:00","notes":"Resolved the two-level column header: Acc (%) falls under genus-level 1-NN probe of unseen species, not seen-species classification or BIN reconstruction. BarcodeBERT (4–4-4) has 78.5 in this cell.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 78.5.","artifact_sha256":"493f9fe70b483780ba76d51ccf217d3ca83539c82b89917fd3ccebe2b6eb831d","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13008329/fullTextXML"},"legacy_id":"b2-barcodebert-2026","legacy_row":{"id":"b2-barcodebert-2026","paper_id":"barcodebert-2026","domain_id":"dna-genomes","task":"unseen-species genus classification","model":"BarcodeBERT (4–4-4)","model_version":"4–4–4","dataset":"DNA barcodes of unseen species","dataset_version":"","split":"1-NN probe","metric":"accuracy","value":"78.5","unit":"percent","uncertainty":"","protocol":"genus-level nearest-neighbor probe on species unseen in training","source_locator":"Table 1, BarcodeBERT (4–4-4) row, unseen-species genus-level 1-NN Acc (%) column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13008329/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-birna-bert-2025","kind":"result","name":"BiRNA-BERT · F1 · extremely long-sequence species classification","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["extremely long RNA species classification"]},"source_ids":["birna-bert-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-birna-bert-2025"}],"attributes":{"printed_value":"0.804","numeric_value":"0.804","metric":"F1","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 2, BiRNA-BERT row, F1 Score column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:38.292Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 2, BiRNA-BERT row, F1 Score column; cell: 0.804","artifact_sha256":"bf7dbc52b6515301c77010c513f13e676c38395ddc82c20310171f518690c152","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12635123/fullTextXML"},"legacy_id":"b2-birna-bert-2025","legacy_row":{"id":"b2-birna-bert-2025","paper_id":"birna-bert-2025","domain_id":"rna-transcriptomes","task":"extremely long RNA species classification","model":"BiRNA-BERT","model_version":"not stated in table","dataset":"extremely long-sequence species classification","dataset_version":"","split":"paper evaluation","metric":"F1","value":"0.804","unit":"fraction","uncertainty":"","protocol":"adaptive tokenization on full-length long RNA sequences","source_locator":"Table 2, BiRNA-BERT row, F1 Score column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12635123/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-cathe2-2025","kind":"result","name":"CATHe2 + ProstT5 · F1 · CATH superfamily benchmark","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["CATH superfamily annotation"]},"source_ids":["cathe2-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-cathe2-2025"}],"attributes":{"printed_value":"82.3","numeric_value":"82.3","metric":"F1","metric_direction":"unknown","unit":"percent","uncertainty":"± 1.3 percentage points","source_locator":"Table 3, ProstT5 full row, F1 score column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:38.366Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 3, ProstT5 full row, F1 score column; cell: 82.3% ± 1.3%","artifact_sha256":"713dbfb6ec1cc1aa85c0543eb93aafa0b45b8873df28053b765dd0a1b6d9b563","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12631783/fullTextXML"},"legacy_id":"b2-cathe2-2025","legacy_row":{"id":"b2-cathe2-2025","paper_id":"cathe2-2025","domain_id":"proteins-complexes","task":"CATH superfamily annotation","model":"CATHe2 + ProstT5","model_version":"full ProstT5","dataset":"CATH superfamily benchmark","dataset_version":"","split":"paper evaluation","metric":"F1","value":"82.3","unit":"percent","uncertainty":"± 1.3 percentage points","protocol":"amino-acid and structural alphabet embedding classifier","source_locator":"Table 3, ProstT5 full row, F1 score column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12631783/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"b2-clathrin-plm-2025","kind":"result","name":"ESM-2 embedding + paper classifier · accuracy · CLA-IND0.6","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["clathrin protein classification"]},"source_ids":["clathrin-plm-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-clathrin-plm-2025"}],"attributes":{"printed_value":"0.916","numeric_value":"0.916","metric":"accuracy","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 2, Independent test / ESM-2 row, ACC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558194+00:00","notes":"Resolved the blank evaluation-strategy cells by their independent-test row group. ESM-2 ACC is 0.916 there; the cross-validation ESM-2 ACC is instead 0.873. This is the paper classifier using embeddings, not a standalone checkpoint.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 0.916.","artifact_sha256":"2edc86b25707c1b737d26117093ce8d856e79cc5d0b335f27c1c341f887f1c7e","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12238356/fullTextXML"},"legacy_id":"b2-clathrin-plm-2025","legacy_row":{"id":"b2-clathrin-plm-2025","paper_id":"clathrin-plm-2025","domain_id":"proteins-complexes","task":"clathrin protein classification","model":"ESM-2 embedding + paper classifier","model_version":"not stated in table","dataset":"CLA-IND0.6","dataset_version":"","split":"independent test","metric":"accuracy","value":"0.916","unit":"fraction","uncertainty":"","protocol":"single-feature ESM-2 embedding comparison","source_locator":"Table 2, Independent test / ESM-2 row, ACC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12238356/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-cobra-rna-binding-2026","kind":"result","name":"ERNIE-RNA + CoBRA · MCC · CoBRA compound-binding test set","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA compound-binding site prediction"]},"source_ids":["cobra-rna-binding-2026"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-cobra-rna-binding-2026"}],"attributes":{"printed_value":"0.657","numeric_value":"0.657","metric":"MCC","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 2, ERNIE-RNA / TCL focal row, MCC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558197+00:00","notes":"Matched ERNIE-RNA jointly with TCL focal loss, then the MCC column. Table 2 explicitly reports test-set models. The cell is 0.657, distinct from AUROC 0.868.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 0.657.","artifact_sha256":"8c6a6f00f5fa5f62acf301a66e9e6fa9ef11c7a05ad9b7447d2ade2ce8eba793","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12790621/fullTextXML"},"legacy_id":"b2-cobra-rna-binding-2026","legacy_row":{"id":"b2-cobra-rna-binding-2026","paper_id":"cobra-rna-binding-2026","domain_id":"rna-transcriptomes","task":"RNA compound-binding site prediction","model":"ERNIE-RNA + CoBRA","model_version":"not stated in table","dataset":"CoBRA compound-binding test set","dataset_version":"","split":"test set","metric":"MCC","value":"0.657","unit":"unitless","uncertainty":"","protocol":"ERNIE-RNA embedding with TCL focal loss","source_locator":"Table 2, ERNIE-RNA / TCL focal row, MCC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12790621/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-codonbert-vaccines-2024","kind":"result","name":"CodonBERT · Spearman rho · flu-vaccine sequences","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["flu-vaccine mRNA property prediction"]},"source_ids":["codonbert-vaccines-2024"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-codonbert-vaccines-2024"}],"attributes":{"printed_value":"0.81","numeric_value":"0.81","metric":"Spearman rho","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 2, CodonBERT row, Flu vaccines Spearman correlation column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558201+00:00","notes":"Matched the CodonBERT row and Flu vaccines column (0.81). The table footnote identifies regression columns as Spearman rank correlation and singles out E. coli as classification; this is not a flu-vaccine accuracy score.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 0.81.","artifact_sha256":"2968073753e6d44feff9c08b131edf23145e95b171434539dddf77bb92847033","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11368176/fullTextXML"},"legacy_id":"b2-codonbert-vaccines-2024","legacy_row":{"id":"b2-codonbert-vaccines-2024","paper_id":"codonbert-vaccines-2024","domain_id":"rna-transcriptomes","task":"flu-vaccine mRNA property prediction","model":"CodonBERT","model_version":"not stated in table","dataset":"flu-vaccine sequences","dataset_version":"","split":"paper evaluation","metric":"Spearman rho","value":"0.81","unit":"unitless","uncertainty":"","protocol":"codon-based model fine-tuned for downstream regression","source_locator":"Table 2, CodonBERT row, Flu vaccines Spearman correlation column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11368176/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-dart-eval-regulatory-2024","kind":"result","name":"DNABERT-2 · accuracy · DART-Eval cCREs versus matched shuffled controls","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["regulatory element identification"]},"source_ids":["dart-eval-regulatory-2024"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-dart-eval-regulatory-2024"}],"attributes":{"printed_value":"0.876","numeric_value":"0.876","metric":"accuracy","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 3 (PDF page 5), DNABERT-2 row, Zero-Shot Accuracy column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558203+00:00","notes":"Inspected the pinned NeurIPS primary PDF table and explanatory text. The DNABERT-2 row reports 0.876 under Zero-Shot Accuracy. The caption defines this as pairwise prioritization of positives over matched controls, distinct from supervised absolute accuracy.","evidence":"DNABERT-2; zero-shot accuracy 0.876; probed absolute/paired 0.847/0.943; fine-tuned absolute/paired 0.913/0.973.","artifact_sha256":"e5aee5b1f7cc6fd961b1d2a131d02cf243b79e091d5e418fbabee7fde9b39b22","retrieval_url":"https://proceedings.neurips.cc/paper_files/paper/2024/file/71998bfc3217ffe1cca1ee084dfadadd-Paper-Datasets_and_Benchmarks_Track.pdf"},"legacy_id":"b2-dart-eval-regulatory-2024","legacy_row":{"id":"b2-dart-eval-regulatory-2024","paper_id":"dart-eval-regulatory-2024","domain_id":"dna-genomes","task":"regulatory element identification","model":"DNABERT-2","model_version":"not stated in table","dataset":"DART-Eval cCREs versus matched shuffled controls","dataset_version":"","split":"paper evaluation","metric":"accuracy","value":"0.876","unit":"fraction","uncertainty":"","protocol":"zero-shot likelihood ranking: higher likelihood for cCRE than matched control","source_locator":"Table 3, DNABERT-2 row, Zero-Shot Accuracy column","source_url":"https://proceedings.neurips.cc/paper_files/paper/2024/file/71998bfc3217ffe1cca1ee084dfadadd-Paper-Datasets_and_Benchmarks_Track.pdf","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-dnabert2-enhancer-2025","kind":"result","name":"DNABERT2-Enhancer · AUC · Liu training dataset","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["enhancer recognition"]},"source_ids":["dnabert2-enhancer-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-dnabert2-enhancer-2025"}],"attributes":{"printed_value":"0.965","numeric_value":"0.965","metric":"AUC","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 4, first-layer DNABERT2-Enhancer row, AUC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558204+00:00","notes":"Resolved the first-layer row group. DNABERT2-Enhancer AUC is 0.965, whereas second-layer AUC is 0.933. The caption explicitly describes 5-fold cross-validation on Liu training data, not an independent held-out test.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 0.965.","artifact_sha256":"d052b80efe7bfc1380994ad28503a5575f04ef940f74d5c9c137cb4ba6827863","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11981215/fullTextXML"},"legacy_id":"b2-dnabert2-enhancer-2025","legacy_row":{"id":"b2-dnabert2-enhancer-2025","paper_id":"dnabert2-enhancer-2025","domain_id":"dna-genomes","task":"enhancer recognition","model":"DNABERT2-Enhancer","model_version":"not stated in table","dataset":"Liu training dataset","dataset_version":"","split":"5-fold cross-validation","metric":"AUC","value":"0.965","unit":"fraction","uncertainty":"","protocol":"first-layer enhancer versus non-enhancer classifier","source_locator":"Table 4, first-layer DNABERT2-Enhancer row, AUC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11981215/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-eden-genomic-classification-2026","kind":"result","name":"DNABERT-2 · MCC · GUE H-CPD","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["human core-promoter classification"]},"source_ids":["eden-genomic-classification-2026"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-eden-genomic-classification-2026"}],"attributes":{"printed_value":"70.52","numeric_value":"70.52","metric":"MCC","metric_direction":"unknown","unit":"percent","uncertainty":null,"source_locator":"Table 5, DNABERT-2 row, H-CPD (MCC) column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:37.531Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 5, DNABERT-2 row, H-CPD (MCC) column; cell: 70.52","artifact_sha256":"38a6e26b3caffe8e021a2b0b672218e783aca9ee42046765e323946813015e65","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12879454/fullTextXML"},"legacy_id":"b2-eden-genomic-classification-2026","legacy_row":{"id":"b2-eden-genomic-classification-2026","paper_id":"eden-genomic-classification-2026","domain_id":"dna-genomes","task":"human core-promoter classification","model":"DNABERT-2","model_version":"not stated in table","dataset":"GUE H-CPD","dataset_version":"","split":"paper evaluation","metric":"MCC","value":"70.52","unit":"percent","uncertainty":"","protocol":"DNABERT-2 comparator in consolidated H-CPD table; rerun provenance not explicit","source_locator":"Table 5, DNABERT-2 row, H-CPD (MCC) column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12879454/","evaluation_origin":"paper_compilation","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-ernie-rna-2025","kind":"result","name":"ERNIE-RNA · binary F1 · bpRNA-new","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary-structure prediction"]},"source_ids":["ernie-rna-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-ernie-rna-2025"}],"attributes":{"printed_value":"0.575","numeric_value":"0.575","metric":"binary F1","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 2, ERNIE-RNA zero-shot row, bpRNA-new F1-Score (binary) column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558206+00:00","notes":"Resolved bpRNA-new as the first three-column dataset group and F1-Score (binary) as its third metric. ERNIE-RNA zero shot is 86M and reports 0.575; RNA3DB-2D F1 is instead 0.542.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 0.575.","artifact_sha256":"0bd1d4b3cbf5d59d452cec4864614947861efcee050ba07e7de395cd90630047","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12627772/fullTextXML"},"legacy_id":"b2-ernie-rna-2025","legacy_row":{"id":"b2-ernie-rna-2025","paper_id":"ernie-rna-2025","domain_id":"rna-transcriptomes","task":"RNA secondary-structure prediction","model":"ERNIE-RNA","model_version":"86M","dataset":"bpRNA-new","dataset_version":"","split":"cross-family test","metric":"binary F1","value":"0.575","unit":"fraction","uncertainty":"","protocol":"zero-shot attention-derived base-pair prediction","source_locator":"Table 2, ERNIE-RNA zero-shot row, bpRNA-new F1-Score (binary) column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12627772/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-esm2-ofs-fitness-2025","kind":"result","name":"ESM2 OFS pseudo-perplexity · Spearman rho · ProteinGym substitutions","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["protein variant fitness prediction"]},"source_ids":["esm2-ofs-fitness-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-esm2-ofs-fitness-2025"}],"attributes":{"printed_value":"0.403","numeric_value":"0.403","metric":"Spearman rho","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Published PDF page 6 (033014-6), Table I, ESM2: OFS PP row, Aggregate mean column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source review, not human review","reviewed_at":"2026-09-16T10:45:41.099916+00:00","notes":"Publisher PDF retrieved through official APS harvest endpoint after direct download returned403. Table I is ProteinGym substitutions, not indels TableII. Last column aggregate mean0.403; separate function categories precede it. This verifies reported score, not experimental reproduction. Comparator rows in this table are sourced from ProteinGym; OFS PP is authors own method.","evidence":"Headers: Activity(43), Binding(14), Expression(17), Organismal fitness(77), Stability(66), Aggregate mean. ESM2:OFS PP row:0.393,0.279,0.397,0.331,0.507,0.403. Verified publisher PDF layout extraction against web-rendered primary PDF table.","artifact_sha256":"085ef646f11b8e5335c4b3d86b15fb6c7bf5edf4a80a8b622753ac69d9991a67","retrieval_url":"https://harvest.aps.org/v2/journals/articles/10.1103/zhx7-hcmm/fulltext"},"legacy_id":"b2-esm2-ofs-fitness-2025","legacy_row":{"id":"b2-esm2-ofs-fitness-2025","paper_id":"esm2-ofs-fitness-2025","domain_id":"proteins-complexes","task":"protein variant fitness prediction","model":"ESM2 OFS pseudo-perplexity","model_version":"not stated in table","dataset":"ProteinGym substitutions","dataset_version":"","split":"aggregate across assays","metric":"Spearman rho","value":"0.403","unit":"unitless","uncertainty":"","protocol":"authors’ zero-shot ESM2 OFS pseudo-perplexity evaluation; aggregate mean across ProteinGym substitution assays","source_locator":"Table I, ESM2: OFS PP row, Aggregate Mean Spearman correlation column","source_url":"https://journals.aps.org/prxlife/pdf/10.1103/zhx7-hcmm","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-fusion-breakpoint-foundation-models-2026","kind":"result","name":"Nucleotide Transformer + NN (middle) · ROC AUC · gene fusion breakpoint DNA sequences","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["gene fusion breakpoint classification"]},"source_ids":["fusion-breakpoint-foundation-models-2026"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-fusion-breakpoint-foundation-models-2026"}],"attributes":{"printed_value":"0.994","numeric_value":"0.994","metric":"ROC AUC","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 2, NT / NN (middle) row, ROC AUC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558209+00:00","notes":"Matched NT jointly with NN (middle) and ROC AUC 0.994 in the full-test-set table. NT with SVM reports 0.995 and is a separate pipeline.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 0.994.","artifact_sha256":"0f4d9de77f1e39cfd2164a20653d86370767da684dc22d17e09f589761abeb5f","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13182013/fullTextXML"},"legacy_id":"b2-fusion-breakpoint-foundation-models-2026","legacy_row":{"id":"b2-fusion-breakpoint-foundation-models-2026","paper_id":"fusion-breakpoint-foundation-models-2026","domain_id":"dna-genomes","task":"gene fusion breakpoint classification","model":"Nucleotide Transformer + NN (middle)","model_version":"not stated in table","dataset":"gene fusion breakpoint DNA sequences","dataset_version":"","split":"full test set","metric":"ROC AUC","value":"0.994","unit":"fraction","uncertainty":"","protocol":"middle embedding with neural-network classifier","source_locator":"Table 2, NT / NN (middle) row, ROC AUC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13182013/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-genomic-tokenizer-selection-2025","kind":"result","name":"Caduceus (character tokens) · MCC · genomic benchmark categories","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["regulatory sequence classification"]},"source_ids":["genomic-tokenizer-selection-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-genomic-tokenizer-selection-2025"}],"attributes":{"printed_value":"0.778","numeric_value":"0.778","metric":"MCC","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 2, Regulatory row, Caduceus (char) MCC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558210+00:00","notes":"Matched Regulatory row with Caduceus (char) column, 0.778. Caption establishes these as MCC summaries by category; model-size row identifies 3.9M parameters. This is an aggregated category result, not a single unspecified split.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 0.778.","artifact_sha256":"0a01c36fdd63f3f6db509777e61c3f87e8a298c810f8aef7974915aaa0655342","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12453675/fullTextXML"},"legacy_id":"b2-genomic-tokenizer-selection-2025","legacy_row":{"id":"b2-genomic-tokenizer-selection-2025","paper_id":"genomic-tokenizer-selection-2025","domain_id":"dna-genomes","task":"regulatory sequence classification","model":"Caduceus (character tokens)","model_version":"3.9M parameter variant","dataset":"genomic benchmark categories","dataset_version":"","split":"paper benchmark summary","metric":"MCC","value":"0.778","unit":"unitless","uncertainty":"","protocol":"task-category MCC across benchmark datasets","source_locator":"Table 2, Regulatory row, Caduceus (char) MCC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12453675/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-gsmformer-ppi-2026","kind":"result","name":"GSMFormer-PPI + ProstT5 · AUROC · paper PPI test set","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["protein-protein interaction prediction"]},"source_ids":["gsmformer-ppi-2026"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-gsmformer-ppi-2026"}],"attributes":{"printed_value":"0.988","numeric_value":"0.988","metric":"AUROC","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 6, ProstT5 embedding row, AUROC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558212+00:00","notes":"Matched ProstT5 embedding row and AUROC column, 0.988. Caption explicitly describes GSMFormer-PPI using embeddings as node features, not standalone ProstT5 prediction.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 0.988.","artifact_sha256":"9b364b5d73d16f2787f93f78f17dbe98b954ab9c2c64c1df960eec2e615eb3b4","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12873117/fullTextXML"},"legacy_id":"b2-gsmformer-ppi-2026","legacy_row":{"id":"b2-gsmformer-ppi-2026","paper_id":"gsmformer-ppi-2026","domain_id":"proteins-complexes","task":"protein-protein interaction prediction","model":"GSMFormer-PPI + ProstT5","model_version":"not stated in table","dataset":"paper PPI test set","dataset_version":"","split":"test set","metric":"AUROC","value":"0.988","unit":"fraction","uncertainty":"","protocol":"ProstT5 embeddings as graph node features","source_locator":"Table 6, ProstT5 embedding row, AUROC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12873117/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-megsite-2025","kind":"result","name":"MegSite + ESM3 · AUC · DNA-129_Test","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["DNA-binding residue prediction"]},"source_ids":["megsite-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-megsite-2025"}],"attributes":{"printed_value":"0.948","numeric_value":"0.948","metric":"AUC","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 2, DNA-129_Test / ESM3 row, AUC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558213+00:00","notes":"Resolved DNA-129_Test row group and ESM3 row. AUC is 0.948; the next numeric cell 0.582 is AP. Caption states an embedding comparison within MegSite.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 0.948.","artifact_sha256":"10d13122331813243d83b84fe6f9294eac7e7c03cde082ebed276191ac41089c","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12496013/fullTextXML"},"legacy_id":"b2-megsite-2025","legacy_row":{"id":"b2-megsite-2025","paper_id":"megsite-2025","domain_id":"proteins-complexes","task":"DNA-binding residue prediction","model":"MegSite + ESM3","model_version":"not stated in table","dataset":"DNA-129_Test","dataset_version":"","split":"independent test","metric":"AUC","value":"0.948","unit":"fraction","uncertainty":"","protocol":"ESM3 multimodal embedding ablation in MegSite","source_locator":"Table 2, DNA-129_Test / ESM3 row, AUC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12496013/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-mrna-lm-2025","kind":"result","name":"mRNA-LM · Spearman rho · mRNA half-life","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["mRNA half-life prediction"]},"source_ids":["mrna-lm-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-mrna-lm-2025"}],"attributes":{"printed_value":"0.696","numeric_value":"0.696","metric":"Spearman rho","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 1, mRNA-LM row, mRNA half-life Spearman column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558214+00:00","notes":"Resolved mRNA half-life column under the Spearman header spanning three tasks. mRNA-LM gives 0.696. Caption identifies average test performance across cross-validation splits; protein-expression AUROC is a different column.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 0.696.","artifact_sha256":"3a23de3c672ec162d13561c483f180a73b550d717256deffdc9099accec205fd","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11962594/fullTextXML"},"legacy_id":"b2-mrna-lm-2025","legacy_row":{"id":"b2-mrna-lm-2025","paper_id":"mrna-lm-2025","domain_id":"rna-transcriptomes","task":"mRNA half-life prediction","model":"mRNA-LM","model_version":"not stated in table","dataset":"mRNA half-life","dataset_version":"","split":"test set across CV splits","metric":"Spearman rho","value":"0.696","unit":"unitless","uncertainty":"","protocol":"average test performance across cross-validation splits","source_locator":"Table 1, mRNA-LM row, mRNA half-life Spearman column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11962594/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-mrnabert-2025","kind":"result","name":"mRNABERT · R-squared · human ultra-long mRNAs","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["translation-efficiency prediction"]},"source_ids":["mrnabert-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-mrnabert-2025"}],"attributes":{"printed_value":"0.669","numeric_value":"0.669","metric":"R-squared","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 2, mRNABERT (3066) row, Human R-squared column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558216+00:00","notes":"Resolved Human group and its R-squared subcolumn. mRNABERT (3066) reports 0.669; Human Spearman is 0.814 and Mouse R-squared is 0.649. Caption specifies ultra-long mRNA translation-efficiency prediction.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 0.669.","artifact_sha256":"ff08ba895b7080446c08a930548b48a0041ae990c222ebb07e6ba7dcaf48ad44","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12644827/fullTextXML"},"legacy_id":"b2-mrnabert-2025","legacy_row":{"id":"b2-mrnabert-2025","paper_id":"mrnabert-2025","domain_id":"rna-transcriptomes","task":"translation-efficiency prediction","model":"mRNABERT","model_version":"3066-nt input","dataset":"human ultra-long mRNAs","dataset_version":"","split":"paper evaluation","metric":"R-squared","value":"0.669","unit":"unitless","uncertainty":"","protocol":"human translation-efficiency regression at 3066-nt input","source_locator":"Table 2, mRNABERT (3066) row, Human R-squared column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12644827/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-mulan-2025","kind":"result","name":"MULAN-ESM2 S · AUC · HumanPPI","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["human protein-protein interaction prediction"]},"source_ids":["mulan-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-mulan-2025"}],"attributes":{"printed_value":"0.717","numeric_value":"0.717","metric":"AUC","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 2, MULAN-ESM2 S row, HumanPPI AUC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558217+00:00","notes":"Resolved the multirow header: HumanPPI uses AUC. MULAN-ESM2 S has 0.717; this is the small-model group, distinct from M and L variants.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 0.717.","artifact_sha256":"771a9a26ebda6f49ea266540e8dd6e6de0cbaef724de818ca6124a5f9c50d350","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12452268/fullTextXML"},"legacy_id":"b2-mulan-2025","legacy_row":{"id":"b2-mulan-2025","paper_id":"mulan-2025","domain_id":"proteins-complexes","task":"human protein-protein interaction prediction","model":"MULAN-ESM2 S","model_version":"small ESM2 backbone","dataset":"HumanPPI","dataset_version":"","split":"paper evaluation","metric":"AUC","value":"0.717","unit":"fraction","uncertainty":"","protocol":"MULAN sequence-structure model based on ESM2 8M","source_locator":"Table 2, MULAN-ESM2 S row, HumanPPI AUC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12452268/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-phylogpn-2025","kind":"result","name":"PhyloGPN · AUROC · ClinVar 3-prime UTR variants","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["ClinVar 3-prime UTR variant classification"]},"source_ids":["phylogpn-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-phylogpn-2025"}],"attributes":{"printed_value":"0.94","numeric_value":"0.94","metric":"AUROC","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 1, 3-prime UTR row, PhyloGPN AUROC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558218+00:00","notes":"Matched 3-prime UTR row and PhyloGPN column (0.94). Caption specifies log-likelihood-ratio predictions of ClinVar classes and explicitly defines each cell as AUROC.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 0.94.","artifact_sha256":"807f3a26cbfa9b5ce238d92164bd523302c67d1c5794b08273c51cca1acd4224","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11908359/fullTextXML"},"legacy_id":"b2-phylogpn-2025","legacy_row":{"id":"b2-phylogpn-2025","paper_id":"phylogpn-2025","domain_id":"dna-genomes","task":"ClinVar 3-prime UTR variant classification","model":"PhyloGPN","model_version":"not stated in table","dataset":"ClinVar 3-prime UTR variants","dataset_version":"","split":"paper evaluation","metric":"AUROC","value":"0.94","unit":"fraction","uncertainty":"","protocol":"log-likelihood-ratio scoring","source_locator":"Table 1, 3-prime UTR row, PhyloGPN AUROC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11908359/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-polya-glm-2025","kind":"result","name":"HyenaDNA · AUC · poly(A) Gene-Gene","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["polyadenylation site detection"]},"source_ids":["polya-glm-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-polya-glm-2025"}],"attributes":{"printed_value":"0.7510","numeric_value":"0.7510","metric":"AUC","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 1, Few-shot HyenaDNA row, G-G AUC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558220+00:00","notes":"Resolved Few-shot group, HyenaDNA row, and G-G subcolumn under AUC (0.7510). IG-G AUC is 0.7541. Caption states averages over five-fold cross-validation and distinguishes negative sampling regions.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 0.7510.","artifact_sha256":"e9ebd53d88837ad8d457881ffee918d2734dcae87d3c5cd03135947b6cf5dbde","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12799945/fullTextXML"},"legacy_id":"b2-polya-glm-2025","legacy_row":{"id":"b2-polya-glm-2025","paper_id":"polya-glm-2025","domain_id":"dna-genomes","task":"polyadenylation site detection","model":"HyenaDNA","model_version":"not stated in table","dataset":"poly(A) Gene-Gene","dataset_version":"","split":"5-fold cross-validation","metric":"AUC","value":"0.7510","unit":"fraction","uncertainty":"","protocol":"few-shot Gene-Gene negative-set comparison","source_locator":"Table 1, Few-shot HyenaDNA row, G-G AUC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12799945/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-rlsite-rna-binding-2025","kind":"result","name":"RLsite · AUC · T18","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA-small-molecule binding-site prediction"]},"source_ids":["rlsite-rna-binding-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-rlsite-rna-binding-2025"}],"attributes":{"printed_value":"0.828","numeric_value":"0.828","metric":"AUC","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 1, RLsite row, T18 AUC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558222+00:00","notes":"Matched RLsite and AUC (0.828). Caption explicitly identifies dataset T18; MCC 0.474 is a different metric.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 0.828.","artifact_sha256":"a50f344e253162ae43f51d7120cfb35a1d0f6114fd8176d760aceb6d05fd95bd","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12417085/fullTextXML"},"legacy_id":"b2-rlsite-rna-binding-2025","legacy_row":{"id":"b2-rlsite-rna-binding-2025","paper_id":"rlsite-rna-binding-2025","domain_id":"rna-transcriptomes","task":"RNA-small-molecule binding-site prediction","model":"RLsite","model_version":"not stated in table","dataset":"T18","dataset_version":"","split":"paper evaluation","metric":"AUC","value":"0.828","unit":"fraction","uncertainty":"","protocol":"RNA language-model plus graph-attention classifier","source_locator":"Table 1, RLsite row, T18 AUC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12417085/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-rnaret-2026","kind":"result","name":"RNAret · F1 · MirTarRAW","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["miRNA-mRNA interaction prediction"]},"source_ids":["rnaret-2026"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-rnaret-2026"}],"attributes":{"printed_value":"0.9622","numeric_value":"0.9622","metric":"F1","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 1, MirTarRAW / 5-mer RNAret row, F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558224+00:00","notes":"Resolved the MirTarRAW section, 5-mer RNAret row, and F1 column (0.9622), distinct from DeepMirTarLeft F1 0.9728. Methods confirm 72/8/20 train/validation/test partition for MirTarRAW.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 0.9622.","artifact_sha256":"e970e7322e07fb3c9d12efd315691cc5de5575a3f2616f4b788614c8c706dd0b","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13111708/fullTextXML"},"legacy_id":"b2-rnaret-2026","legacy_row":{"id":"b2-rnaret-2026","paper_id":"rnaret-2026","domain_id":"rna-transcriptomes","task":"miRNA-mRNA interaction prediction","model":"RNAret","model_version":"5-mer","dataset":"MirTarRAW","dataset_version":"","split":"held-out test","metric":"F1","value":"0.9622","unit":"fraction","uncertainty":"","protocol":"5-mer RNAret classifier; 72/8/20 train/validation/test split","source_locator":"Table 1, MirTarRAW / 5-mer RNAret row, F1 column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13111708/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-spin-protein-function-2026","kind":"result","name":"SPIN + ESM2-35M · F1 macro-weighted · TRX","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["protein function annotation"]},"source_ids":["spin-protein-function-2026"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-spin-protein-function-2026"}],"attributes":{"printed_value":"0.796","numeric_value":"0.796","metric":"F1 macro-weighted","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 1, ESM2-35M Test row, F1_m-w column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558225+00:00","notes":"Resolved Test group and macro-weighted F1 subcolumn (0.796) for frozen ESM2-35M in SPIN. Test weighted accuracy is 0.798. Methods define inverse-frequency class weighting for macro-weighted F1.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 0.796.","artifact_sha256":"9701843e93bf7fa3ead71e19693fb07d483f1022379871adfb04486783722a9d","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12970593/fullTextXML"},"legacy_id":"b2-spin-protein-function-2026","legacy_row":{"id":"b2-spin-protein-function-2026","paper_id":"spin-protein-function-2026","domain_id":"proteins-complexes","task":"protein function annotation","model":"SPIN + ESM2-35M","model_version":"ESM2-35M frozen","dataset":"TRX","dataset_version":"","split":"test set","metric":"F1 macro-weighted","value":"0.796","unit":"fraction","uncertainty":"","protocol":"frozen ESM2-35M backbone in SPIN","source_locator":"Table 1, ESM2-35M Test row, F1_m-w column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12970593/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-structure-informed-plm-2025","kind":"result","name":"structure-informed pLM · AUROC · variant-effects benchmark","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["protein variant-effect classification"]},"source_ids":["structure-informed-plm-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-structure-informed-plm-2025"}],"attributes":{"printed_value":".803","numeric_value":"0.803","metric":"AUROC","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"PMC12068927 HTML, Table4, AA+SS+RSA+CM row, AUROC column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source review, not human review","reviewed_at":"2026-09-16T10:45:41.099916+00:00","notes":"Full-text HTML succeeds although EuropePMC XMLreturned404. Row is mutation-site variables AA+SS+RSA+CM, not neighbouring environment variant. AUROC .803 is numerically equivalent to preserved legacy0.803. Source check, not experimental reproduction; do not claim original source printed leading zero.","evidence":"Table4 headers: Type, Variable(s), Spearman rho, AUROC, AUPRC. Parsed HTML row: AA+SS+RSA+CM | .552 | .803 | .792. Primary web rendering independently confirms columns.","artifact_sha256":"76082e1cd992d2c09c38f86d05aba575cc76c5022b53a297123b713bb1ce9267","retrieval_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12068927/"},"legacy_id":"b2-structure-informed-plm-2025","legacy_row":{"id":"b2-structure-informed-plm-2025","paper_id":"structure-informed-plm-2025","domain_id":"proteins-complexes","task":"protein variant-effect classification","model":"structure-informed pLM","model_version":"not stated in table","dataset":"variant-effects benchmark","dataset_version":"","split":"paper evaluation","metric":"AUROC","value":"0.803","unit":"fraction","uncertainty":"","protocol":"combined amino-acid, secondary structure, solvent accessibility and contact-map scoring","source_locator":"Table 4, AA+SS+RSA+CM row, AUROC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12068927/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"barcodebert-2026","kind":"source","name":"BarcodeBERT: transformers for biodiversity analyses","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["dna-genomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13008329/","version":"journal full text in PMC","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.1093/bioadv/vbag054","publication_status":"peer_reviewed","year":2026,"artifact_sha256":"493f9fe70b483780ba76d51ccf217d3ca83539c82b89917fd3ccebe2b6eb831d","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13008329/fullTextXML","artifact_retrieved_at":"2026-09-16T10:38:57.558051+00:00","legacy_paper":{"id":"barcodebert-2026","title":"BarcodeBERT: transformers for biodiversity analyses","year":2026,"publication_status":"peer_reviewed","version":"journal full text in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13008329/","primary_domain":"dna-genomes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.1093/bioadv/vbag054","notes":"Numeric result checked against Table 1 in primary full-text XML; journal/source: Bioinformatics Advances."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"birna-bert-2025","kind":"source","name":"BiRNA-BERT allows efficient RNA language modeling with adaptive tokenization","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["rna-transcriptomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12635123/","version":"journal full text in PMC","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.1038/s42003-025-08982-0","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"bf7dbc52b6515301c77010c513f13e676c38395ddc82c20310171f518690c152","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12635123/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:38.292Z","legacy_paper":{"id":"birna-bert-2025","title":"BiRNA-BERT allows efficient RNA language modeling with adaptive tokenization","year":2025,"publication_status":"peer_reviewed","version":"journal full text in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12635123/","primary_domain":"rna-transcriptomes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.1038/s42003-025-08982-0","notes":"Numeric result checked against Table 2 in primary full-text XML; journal/source: Communications Biology."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"boltz-stereochemistry-2025","kind":"source","name":"Improving Stereochemical Limitations in Protein–Ligand Complex Structure Prediction","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12658688/","version":"version of record","retrieved_at":"2026-09-15T23:25:00Z","doi":"10.1021/acsomega.5c07675","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"78a77b9a0ab8bfa371f5b9baef3f443f4590d6e71cf864d67e90e9ebdfa7fc1b","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12658688/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:16.555674+00:00","legacy_paper":{"id":"boltz-stereochemistry-2025","title":"Improving Stereochemical Limitations in Protein–Ligand Complex Structure Prediction","year":2025,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12658688/","primary_domain":"molecular-interactions","retrieved_utc":"2026-09-15T23:25:00Z","notes":"Primary full text via Europe PMC XML; venue: ACS Omega; PMC ID: PMC12658688.","doi":"10.1021/acsomega.5c07675"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"boltz1-2025","kind":"source","name":"Boltz-1 Democratizing Biomolecular Interaction Modeling","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11601547/","version":"PMC archival version PMC11601547.4","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.1101/2024.11.19.624167","publication_status":"preprint","year":2025,"artifact_sha256":"1ebf712314d9a1c678ded989cc95a0c00c0331e5ad8c9f63194bc9780971d214","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11601547/fullTextXML","artifact_retrieved_at":"2026-09-16T10:44:03.424864+00:00","legacy_paper":{"id":"boltz1-2025","title":"Boltz-1 Democratizing Biomolecular Interaction Modeling","year":2025,"publication_status":"preprint","version":"PMC archival version PMC11601547.4","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11601547/","primary_domain":"molecular-interactions","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: bioRxiv; PMC ID: PMC11601547.","doi":"10.1101/2024.11.19.624167"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"bpfold-2025","kind":"source","name":"Deep generalizable prediction of RNA secondary structure via base pair motif energy","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["rna-transcriptomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12216785/","version":"version of record","retrieved_at":"2026-09-15T23:25:00Z","doi":"10.1038/s41467-025-60048-1","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"976218bd172998a1a6e7ed1609ecb8cb2ee380fb48a8dc7b25bc05ea8b0a49af","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12216785/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:16.502000+00:00","legacy_paper":{"id":"bpfold-2025","title":"Deep generalizable prediction of RNA secondary structure via base pair motif energy","year":2025,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12216785/","primary_domain":"rna-transcriptomes","retrieved_utc":"2026-09-15T23:25:00Z","notes":"Primary full text via Europe PMC XML; venue: Nature Communications; PMC ID: PMC12216785.","doi":"10.1038/s41467-025-60048-1"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"cammiq-2022","kind":"source","name":"Strain level microbial detection and quantification with applications to single cell metagenomics","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC9616933/","version":"version of record","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.1038/s41467-022-33869-7","publication_status":"peer_reviewed","year":2022,"artifact_sha256":"f0939647ed3de995d58254f79472a612c21b0e1b2560a82783302aa1a148dde3","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC9616933/fullTextXML","artifact_retrieved_at":"2026-09-16T10:44:03.408237+00:00","legacy_paper":{"id":"cammiq-2022","title":"Strain level microbial detection and quantification with applications to single cell metagenomics","year":2022,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC9616933/","primary_domain":"microbes-communities","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: Nature Communications; PMC ID: PMC9616933.","doi":"10.1038/s41467-022-33869-7"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"catalog-baseline-kraken2","kind":"baseline","name":"Kraken2","description":"","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":["catalog-source-kraken2"],"links":[{"relation":"model","target_id":"catalog-model-kraken2"},{"relation":"applicable_to","target_id":"catalog-task-heldout-clade"},{"relation":"applicable_to","target_id":"catalog-task-phage-pathogen-reads"}],"attributes":{"baseline_type":"established_method","applicability":"proposed","requirements":"Public classifier; database build/version must be pinned separately.","missing_metadata":{"exact_protocol":"not_yet_extracted"}}} {"id":"catalog-baseline-scvi","kind":"baseline","name":"scVI","description":"","status":"discovered","facets":{"areas":["cells-tissues"]},"source_ids":["catalog-source-scvi"],"links":[{"relation":"model","target_id":"catalog-model-scvi"},{"relation":"applicable_to","target_id":"catalog-task-cell-reference-mapping"},{"relation":"applicable_to","target_id":"catalog-task-cell-batch-integration"}],"attributes":{"baseline_type":"established_method","applicability":"proposed","requirements":"Public software; train a task-specific model on the permitted split.","missing_metadata":{"exact_protocol":"not_yet_extracted"}}} {"id":"catalog-baseline-vina","kind":"baseline","name":"AutoDock Vina","description":"","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":["catalog-source-vina"],"links":[{"relation":"model","target_id":"catalog-model-vina"},{"relation":"applicable_to","target_id":"catalog-task-ligand-pose"}],"attributes":{"baseline_type":"established_method","applicability":"proposed","requirements":"Public docking software; receptor and ligand preparation required.","missing_metadata":{"exact_protocol":"not_yet_extracted"}}} {"id":"catalog-model-alphafold-3-server","kind":"model","name":"AlphaFold 3 Server","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["molecular-interactions"],"method_types":["foundation model"]},"source_ids":["catalog-source-alphafold-3-server"],"links":[{"relation":"uses_model","target_id":"discovery-model-alphafold-3"}],"attributes":{"entity_level":"service","version":"hosted server","reported_name":"AlphaFold 3 Server","access":"Manual, non-commercial server access; output terms restrict automated docking combinations.","method_type":"foundation model","historical_missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"AlphaFold Server is Google DeepMind’s hosted interface to AlphaFold 3. It predicts molecular complex structures without a local installation and returns confidence estimates. Its service limits, supported inputs and output terms are separate from the downloadable model.","summary_source_ids":["evidence-alphafold-server-faq","evidence-alphafold-server-terms"],"summary_source_locator":"FAQ: supported molecules, job size, outputs; Terms: Overview","sections":[{"title":"Joint structure prediction","body":"Sequence, chemical and evolutionary features feed a Pairformer, which builds representations of individual tokens and their relationships. A diffusion module then predicts atomic coordinates. Separate heads estimate confidence. The paper describes 48 Pairformer blocks; the architecture models complexes jointly rather than treating every partner as a separately folded structure.","source_ids":["evidence-alphafold-paper"],"source_locator":"Main text: Model architecture; Fig. 1d and Fig. 2"},{"title":"What the server adds","body":"The service handles the input preparation and presents ranked structures and confidence outputs for download. The current FAQ supports custom protein MSAs and templates, as well as ligands specified by CCD code. These options must be recorded when interpreting a result; a server run is not automatically the same configuration as the original paper.","source_ids":["evidence-alphafold-server-faq"],"source_locator":"What structure templates and MSA are used?; How do I add a ligand using its CCD code?; How many predictions are returned?"},{"title":"Service identity and reproducibility","body":"The FAQ states that the server and released AlphaFold 3 use the same weights and equivalent model code. Their genetic-search implementations can nevertheless produce different alignments, so matching weights does not make every run equivalent. The hosted checkpoint digest is not disclosed, and compiler changes can affect exact repeatability. Preserve job inputs, seeds and outputs when citing an evaluation.","source_ids":["evidence-alphafold-server-faq"],"source_locator":"Why might Multiple Sequence Alignments differ between AlphaFold 3 and AlphaFold Server?; job repeatability and seed questions"}],"facts":[{"label":"Model type","value":"Hosted AlphaFold 3 structure-prediction service.","status":"source_checked","source_ids":["evidence-alphafold-server-terms"],"source_locator":"Overview"},{"label":"Architecture","value":"A 48-block Pairformer builds token and pair representations; a diffusion module predicts atomic coordinates and separate heads estimate confidence.","status":"source_checked","source_ids":["evidence-alphafold-paper"],"source_locator":"Model architecture; Fig. 1d and Fig. 2"},{"label":"Known versions","value":"The FAQ states that the released AlphaFold 3 and server use the same weights and equivalent model code. It does not supply a public digest identifying the hosted checkpoint.","status":"source_checked","source_ids":["evidence-alphafold-server-faq"],"source_locator":"Why might Multiple Sequence Alignments differ between AlphaFold 3 and AlphaFold Server?"},{"label":"Inputs","value":"Protein, DNA and RNA sequences, supported modifications, ions and ligands. Current FAQ also permits additional ligands by CCD code and custom protein MSAs/templates.","status":"source_checked","source_ids":["evidence-alphafold-server-faq"],"source_locator":"Supported molecule types; CCD code; template and MSA questions"},{"label":"Outputs","value":"Five predictions per seed, downloadable structures and confidence JSON; the top-ranked prediction is shown in the interface.","status":"source_checked","source_ids":["evidence-alphafold-server-faq"],"source_locator":"How many predictions are returned?; downloaded JSON files"},{"label":"Training data","value":"Uses AlphaFold 3. The paper describes PDB-based training and distillation; the server FAQ does not provide a separate hosted-checkpoint training inventory.","status":"source_checked","source_ids":["evidence-alphafold-paper","evidence-alphafold-server-faq"],"source_locator":"Paper Methods: Training regime; FAQ: templates and MSA"},{"label":"Training cutoff","value":"The paper gives 2021-09-30 for the standard model’s structural training data. The FAQ does not publish a complete dated training manifest for the hosted deployment; its template-search cutoff is a separate input setting.","status":"source_checked","source_ids":["evidence-alphafold-paper","evidence-alphafold-server-faq"],"source_locator":"Paper Methods: Training regime; FAQ: template and MSA settings and comparison with released AlphaFold 3"},{"label":"Context limits","value":"5,000 tokens per job. Residues, nucleotides and molecular atoms count differently; this is a hosted-service limit.","status":"source_checked","source_ids":["evidence-alphafold-server-faq"],"source_locator":"What is the maximum job size allowed?"},{"label":"Access","value":"Google-account web service for non-commercial use. FAQ reports 30 jobs per day as checked on 2026-09-16; quotas may change.","status":"source_checked","source_ids":["evidence-alphafold-server-faq","evidence-alphafold-server-terms"],"source_locator":"How many jobs can I run?; Terms: Key things to know"},{"label":"Code licence","value":"The service is governed by its Terms of Service. Apache-2.0 applies to the separate downloadable inference code, not to ownership or licensing of the hosted service.","status":"source_checked","source_ids":["evidence-alphafold-server-terms","evidence-alphafold-license"],"source_locator":"Terms: Overview; separate implementation LICENSE"},{"label":"Weights licence","value":"Hosted use does not supply a weights licence. Downloadable model parameters are a separate offering under their own terms.","status":"inapplicable","source_ids":["evidence-alphafold-server-terms","evidence-alphafold-weights-terms-of-use"],"source_locator":"Service Terms; Model Parameters Terms"},{"label":"Output terms","value":"Non-commercial use restrictions apply. Terms prohibit use with automated protein–ligand/peptide interaction prediction systems and training similar structure-prediction models; downstream notices are required.","status":"source_checked","source_ids":["evidence-alphafold-server-output-terms"],"source_locator":"Use restrictions"},{"label":"Parameters","value":"The hosted checkpoint’s total parameter count is not supplied in the inspected FAQ. No number is inferred from the model name.","status":"unreported","source_ids":["evidence-alphafold-server-faq"],"source_locator":"FAQ: model identity and access; no total parameter count"}],"strengths":[{"text":"Provides a hosted route to predictions and downloadable confidence outputs without maintaining the local inference installation.","source_ids":["evidence-alphafold-server-faq","evidence-alphafold-server-terms"],"source_locator":"Terms: Overview; FAQ: downloaded outputs"},{"text":"Current input controls expose custom protein MSAs and templates, which helps users document those inputs.","source_ids":["evidence-alphafold-server-faq"],"source_locator":"What structure templates and MSA are used?"}],"limitations":[{"text":"Predictions can contain incorrect chirality, atomic clashes or spurious structure in disordered regions. Confidence and structural plausibility need separate inspection.","source_ids":["evidence-alphafold-paper"],"source_locator":"Model limitations; Fig. 5"},{"text":"Sampled structures are not a calibrated solution-state ensemble. Prediction confidence does not establish binding affinity or experimental function.","source_ids":["evidence-alphafold-paper"],"source_locator":"Model limitations: dynamics and conformational states; confidence outputs are structure-quality estimates"},{"text":"Usage quotas and server terms constrain access and downstream reuse; the service is not an unrestricted batch prediction API.","source_ids":["evidence-alphafold-server-faq","evidence-alphafold-server-terms","evidence-alphafold-server-output-terms"],"source_locator":"Daily quota; Terms and Output Terms: use restrictions"},{"text":"A hosted result cannot establish performance for an unspecified local checkpoint or paper evaluation variant.","source_ids":["evidence-alphafold-server-faq","evidence-alphafold-paper"],"source_locator":"FAQ reproducibility; paper Methods: separate evaluation variants"}],"diagram":{"title":"AlphaFold Server workflow","steps":["Specify molecules and input settings","Hosted preparation of features","AlphaFold 3 structure prediction","Rank samples and inspect confidence","Download structures and job records"],"caption":"Conceptual service workflow. The deployed checkpoint is not pinned by the public FAQ.","source_ids":["evidence-alphafold-server-faq"],"source_locator":"Input, templates/MSAs, returned predictions and download questions"},"coverage":"reviewed","gaps":["Exact hosted checkpoint digest and parameter count remain unreported in the inspected public FAQ.","Service quotas and input options are time-dependent; checked on 2026-09-16."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Read primary paper XML, pinned official repository documentation and licences. Reviewed public server FAQ separately. No model run, independent performance replication or human review. A second automated reviewer checked the AlphaFold source claims and service/model distinction; this is not human review or experimental reproduction."}}}} {"id":"catalog-model-alphagenome","kind":"model","name":"AlphaGenome","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["dna-genomes"],"method_types":["foundation model"]},"source_ids":["catalog-source-alphagenome"],"links":[],"attributes":{"entity_level":"family","version":"API / released weights","reported_name":"AlphaGenome","access":"Rate-limited, non-commercial API requires a key. Downloadable weights require accepting non-commercial model terms; local inference recommends an H100 GPU.","method_type":"foundation model","historical_missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"AlphaGenome predicts regulatory activity and variant effects from long DNA sequences, with outputs for expression, splicing, chromatin and contact maps.","summary_source_ids":["evidence-official-578439da3f3a3a476b7f","evidence-official-56e5abfb5f12f1cd3b20","evidence-official-bc0327ef2ea6107e1773"],"summary_source_locator":"Paper: Unifying DNA sequence-to-function model and Discussion; README.md: Model weights and licensing; Supplementary Methods: Model (p.9), ENCODE RNA-seq Data (p.3), Contact maps","sections":[{"title":"How it works","body":"AlphaGenome progressively downsamples DNA with convolutional blocks, then uses a transformer tower and pairwise interaction blocks to represent long-range context. A U-Net-style decoder restores sequence resolution using skip connections. Modality-specific heads predict one-dimensional genomic tracks, splicing outputs and two-dimensional contact maps.","source_ids":["evidence-official-578439da3f3a3a476b7f","evidence-official-56e5abfb5f12f1cd3b20","evidence-official-bc0327ef2ea6107e1773"],"source_locator":"Paper: Unifying DNA sequence-to-function model and Discussion; README.md: Model weights and licensing; Supplementary Methods: Model (p.9), ENCODE RNA-seq Data (p.3), Contact maps"},{"title":"Versions and reproducibility","body":"The paper distinguishes fold-specific evaluation models, all-fold teachers and distilled students; a family name alone does not choose one of these configurations. Up to 1 million base pairs; single-base outputs for the modalities described in the README.","source_ids":["evidence-official-578439da3f3a3a476b7f","evidence-official-56e5abfb5f12f1cd3b20","evidence-official-bc0327ef2ea6107e1773"],"source_locator":"Paper: Unifying DNA sequence-to-function model and Discussion; README.md: Model weights and licensing; Supplementary Methods: Model (p.9), ENCODE RNA-seq Data (p.3), Contact maps"}],"facts":[{"label":"Model type","value":"Sequence-to-function convolutional/transformer model","status":"source_checked","source_ids":["evidence-official-578439da3f3a3a476b7f","evidence-official-56e5abfb5f12f1cd3b20","evidence-official-bc0327ef2ea6107e1773"],"source_locator":"Paper: Unifying DNA sequence-to-function model and Discussion; README.md: Model weights and licensing; Supplementary Methods: Model (p.9), ENCODE RNA-seq Data (p.3), Contact maps"},{"label":"Architecture","value":"U-Net-inspired sequence backbone combining convolutional local processing with transformer blocks for longer-range interactions; one-dimensional track heads and two-dimensional contact-map representations.","status":"source_checked","source_ids":["evidence-official-578439da3f3a3a476b7f","evidence-official-56e5abfb5f12f1cd3b20","evidence-official-bc0327ef2ea6107e1773"],"source_locator":"Paper: Unifying DNA sequence-to-function model and Discussion; README.md: Model weights and licensing; Supplementary Methods: Model (p.9), ENCODE RNA-seq Data (p.3), Contact maps"},{"label":"Inputs","value":"DNA sequence and optional variant information.","status":"source_checked","source_ids":["evidence-official-578439da3f3a3a476b7f","evidence-official-56e5abfb5f12f1cd3b20","evidence-official-bc0327ef2ea6107e1773"],"source_locator":"Paper: Unifying DNA sequence-to-function model and Discussion; README.md: Model weights and licensing; Supplementary Methods: Model (p.9), ENCODE RNA-seq Data (p.3), Contact maps"},{"label":"Outputs","value":"Predicted expression, splicing, chromatin features and contact maps.","status":"source_checked","source_ids":["evidence-official-578439da3f3a3a476b7f","evidence-official-56e5abfb5f12f1cd3b20","evidence-official-bc0327ef2ea6107e1773"],"source_locator":"Paper: Unifying DNA sequence-to-function model and Discussion; README.md: Model weights and licensing; Supplementary Methods: Model (p.9), ENCODE RNA-seq Data (p.3), Contact maps"},{"label":"Parameters","value":"Approximately 450M trainable parameters, including encoder, sequence transformer, pairwise blocks, decoder and prediction heads.","status":"source_checked","source_ids":["evidence-official-578439da3f3a3a476b7f","evidence-official-56e5abfb5f12f1cd3b20","evidence-official-bc0327ef2ea6107e1773"],"source_locator":"Paper: Unifying DNA sequence-to-function model and Discussion; README.md: Model weights and licensing; Supplementary Methods: Model (p.9), ENCODE RNA-seq Data (p.3), Contact maps"},{"label":"Known versions","value":"The paper distinguishes fold-specific evaluation models, all-fold teachers and distilled students; a family name alone does not choose one of these configurations.","status":"source_checked","source_ids":["evidence-official-578439da3f3a3a476b7f","evidence-official-56e5abfb5f12f1cd3b20","evidence-official-bc0327ef2ea6107e1773"],"source_locator":"Paper: Unifying DNA sequence-to-function model and Discussion; README.md: Model weights and licensing; Supplementary Methods: Model (p.9), ENCODE RNA-seq Data (p.3), Contact maps"},{"label":"Training data","value":"Human and mouse molecular datasets; separate fold-specific models for held-out reference-interval evaluation and all-fold teachers for student distillation.","status":"source_checked","source_ids":["evidence-official-578439da3f3a3a476b7f","evidence-official-56e5abfb5f12f1cd3b20","evidence-official-bc0327ef2ea6107e1773"],"source_locator":"Paper: Unifying DNA sequence-to-function model and Discussion; README.md: Model weights and licensing; Supplementary Methods: Model (p.9), ENCODE RNA-seq Data (p.3), Contact maps"},{"label":"Training cutoff","value":"ENCODE RNA-seq and chromatin metadata were downloaded 9–17 January 2025; the contact-map source was accessed 4 March 2021. These are component retrieval dates, not one universal latest-experiment cutoff.","status":"source_checked","source_ids":["evidence-official-578439da3f3a3a476b7f","evidence-official-56e5abfb5f12f1cd3b20","evidence-official-bc0327ef2ea6107e1773"],"source_locator":"Paper: Unifying DNA sequence-to-function model and Discussion; README.md: Model weights and licensing; Supplementary Methods: Model (p.9), ENCODE RNA-seq Data (p.3), Contact maps"},{"label":"Context limits","value":"Up to 1 million base pairs; single-base outputs for the modalities described in the README.","status":"source_checked","source_ids":["evidence-official-578439da3f3a3a476b7f","evidence-official-56e5abfb5f12f1cd3b20","evidence-official-bc0327ef2ea6107e1773"],"source_locator":"Paper: Unifying DNA sequence-to-function model and Discussion; README.md: Model weights and licensing; Supplementary Methods: Model (p.9), ENCODE RNA-seq Data (p.3), Contact maps"},{"label":"Weights licence","value":"Non-commercial AlphaGenome model terms; not the Apache licence covering source code.","status":"source_checked","source_ids":["evidence-official-578439da3f3a3a476b7f","evidence-official-56e5abfb5f12f1cd3b20","evidence-official-bc0327ef2ea6107e1773"],"source_locator":"Paper: Unifying DNA sequence-to-function model and Discussion; README.md: Model weights and licensing; Supplementary Methods: Model (p.9), ENCODE RNA-seq Data (p.3), Contact maps"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/google-deepmind/alphagenome_research","status":"source_checked","source_ids":["evidence-official-578439da3f3a3a476b7f","evidence-official-56e5abfb5f12f1cd3b20","evidence-official-bc0327ef2ea6107e1773"],"source_locator":"Paper: Unifying DNA sequence-to-function model and Discussion; README.md: Model weights and licensing; Supplementary Methods: Model (p.9), ENCODE RNA-seq Data (p.3), Contact maps"},{"label":"Code licence","value":"Apache-2.0","status":"source_checked","source_ids":["evidence-official-c3d7a64294dcad7c6405"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"Combines several molecular readouts in one sequence model, with variant scoring utilities.","source_ids":["evidence-official-578439da3f3a3a476b7f","evidence-official-56e5abfb5f12f1cd3b20","evidence-official-bc0327ef2ea6107e1773"],"source_locator":"Paper: Unifying DNA sequence-to-function model and Discussion; README.md: Model weights and licensing; Supplementary Methods: Model (p.9), ENCODE RNA-seq Data (p.3), Contact maps"}],"limitations":[{"text":"The paper identifies remaining challenges beyond 100kb, context-specific variant effects and non-coding genes. Species coverage is human/mouse and personal-genome prediction was not benchmarked in the study.","source_ids":["evidence-official-578439da3f3a3a476b7f","evidence-official-56e5abfb5f12f1cd3b20","evidence-official-bc0327ef2ea6107e1773"],"source_locator":"Paper: Unifying DNA sequence-to-function model and Discussion; README.md: Model weights and licensing; Supplementary Methods: Model (p.9), ENCODE RNA-seq Data (p.3), Contact maps"}],"diagram":{"title":"AlphaGenome workflow","steps":["DNA and species identifier","Convolutional sequence encoder","Transformer and pairwise blocks","Decoder with skip connections","Track, splicing and contact-map heads"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-578439da3f3a3a476b7f","evidence-official-56e5abfb5f12f1cd3b20","evidence-official-bc0327ef2ea6107e1773"],"source_locator":"Paper: Unifying DNA sequence-to-function model and Discussion; README.md: Model weights and licensing; Supplementary Methods: Model (p.9), ENCODE RNA-seq Data (p.3), Contact maps"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}}} {"id":"catalog-model-boltz-2","kind":"model","name":"Boltz-2","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["molecular-interactions"],"method_types":["foundation model"]},"source_ids":["catalog-source-boltz-2"],"links":[{"relation":"variant_of","target_id":"discovery-model-boltz"}],"attributes":{"entity_level":"family","version":"released weights","reported_name":"Boltz-2","access":"Public MIT code and weights; substantial compute required.","method_type":"foundation model","historical_missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"Boltz predicts biomolecular complex structures; Boltz-2 also predicts binding affinity.","summary_source_ids":["evidence-official-f780521ccebbb0691f9e","evidence-official-5a9e55abdf288d9dd8da","evidence-official-39a15ed8072aeea14a22","evidence-official-734117ed8501eb26fefc","evidence-official-7cb8cb7f091536b92c11","evidence-official-bd6ec1f819e22f075c32"],"summary_source_locator":"src/boltz/model/models/boltz2.py: Boltz2.__init__, forward, PairformerModule, AtomDiffusion and AffinityModule; docs/prediction.md: affinity outputs; Boltz-2 report Sections 2 Data, 3 Architecture, 4 Training and 6 Limitations","sections":[{"title":"How it works","body":"Boltz-2 first encodes the molecular inputs, alignments and optional templates into token and pair features. A Pairformer trunk updates those features and conditions atom-coordinate diffusion to generate a complex. Separate confidence and affinity modules assess the prediction; affinity classification and regression outputs answer different questions.","source_ids":["evidence-official-f780521ccebbb0691f9e","evidence-official-5a9e55abdf288d9dd8da","evidence-official-39a15ed8072aeea14a22","evidence-official-734117ed8501eb26fefc","evidence-official-7cb8cb7f091536b92c11","evidence-official-bd6ec1f819e22f075c32"],"source_locator":"src/boltz/model/models/boltz2.py: Boltz2.__init__, forward, PairformerModule, AtomDiffusion and AffinityModule; docs/prediction.md: affinity outputs; Boltz-2 report Sections 2 Data, 3 Architecture, 4 Training and 6 Limitations"},{"title":"Versions and reproducibility","body":"Boltz-1 and Boltz-2 are distinct released generations; the catalogue does not select an evaluated checkpoint. The Boltz-2 report describes training crops up to 768 tokens. This is a training-crop size rather than a universal inference maximum; affinity additionally uses a pocket crop.","source_ids":["evidence-official-f780521ccebbb0691f9e","evidence-official-5a9e55abdf288d9dd8da","evidence-official-39a15ed8072aeea14a22","evidence-official-734117ed8501eb26fefc","evidence-official-7cb8cb7f091536b92c11","evidence-official-bd6ec1f819e22f075c32"],"source_locator":"src/boltz/model/models/boltz2.py: Boltz2.__init__, forward, PairformerModule, AtomDiffusion and AffinityModule; docs/prediction.md: affinity outputs; Boltz-2 report Sections 2 Data, 3 Architecture, 4 Training and 6 Limitations"}],"facts":[{"label":"Model type","value":"Biomolecular structure and affinity predictor","status":"source_checked","source_ids":["evidence-official-f780521ccebbb0691f9e","evidence-official-5a9e55abdf288d9dd8da","evidence-official-39a15ed8072aeea14a22","evidence-official-734117ed8501eb26fefc","evidence-official-7cb8cb7f091536b92c11","evidence-official-bd6ec1f819e22f075c32"],"source_locator":"src/boltz/model/models/boltz2.py: Boltz2.__init__, forward, PairformerModule, AtomDiffusion and AffinityModule; docs/prediction.md: affinity outputs; Boltz-2 report Sections 2 Data, 3 Architecture, 4 Training and 6 Limitations"},{"label":"Architecture","value":"Molecular input embeddings, MSA and optional template modules build single-token and pair representations. Pairformer blocks refine these features; atom-coordinate diffusion generates structures, with separate confidence and binding-affinity modules.","status":"source_checked","source_ids":["evidence-official-f780521ccebbb0691f9e","evidence-official-5a9e55abdf288d9dd8da","evidence-official-39a15ed8072aeea14a22","evidence-official-734117ed8501eb26fefc","evidence-official-7cb8cb7f091536b92c11","evidence-official-bd6ec1f819e22f075c32"],"source_locator":"src/boltz/model/models/boltz2.py: Boltz2.__init__, forward, PairformerModule, AtomDiffusion and AffinityModule; docs/prediction.md: affinity outputs; Boltz-2 report Sections 2 Data, 3 Architecture, 4 Training and 6 Limitations"},{"label":"Inputs","value":"Protein, nucleic-acid and ligand specifications in prediction input files.","status":"source_checked","source_ids":["evidence-official-f780521ccebbb0691f9e","evidence-official-5a9e55abdf288d9dd8da","evidence-official-39a15ed8072aeea14a22","evidence-official-734117ed8501eb26fefc","evidence-official-7cb8cb7f091536b92c11","evidence-official-bd6ec1f819e22f075c32"],"source_locator":"src/boltz/model/models/boltz2.py: Boltz2.__init__, forward, PairformerModule, AtomDiffusion and AffinityModule; docs/prediction.md: affinity outputs; Boltz-2 report Sections 2 Data, 3 Architecture, 4 Training and 6 Limitations"},{"label":"Outputs","value":"Predicted complex structures and, for supported Boltz-2 inputs, binding-affinity predictions.","status":"source_checked","source_ids":["evidence-official-f780521ccebbb0691f9e","evidence-official-5a9e55abdf288d9dd8da","evidence-official-39a15ed8072aeea14a22","evidence-official-734117ed8501eb26fefc","evidence-official-7cb8cb7f091536b92c11","evidence-official-bd6ec1f819e22f075c32"],"source_locator":"src/boltz/model/models/boltz2.py: Boltz2.__init__, forward, PairformerModule, AtomDiffusion and AffinityModule; docs/prediction.md: affinity outputs; Boltz-2 report Sections 2 Data, 3 Architecture, 4 Training and 6 Limitations"},{"label":"Parameters","value":"A complete parameter total is not stated in the reviewed Boltz-2 architecture report or model constructor; structure, confidence and affinity are separate modules.","status":"unreported","source_ids":["evidence-official-f780521ccebbb0691f9e","evidence-official-5a9e55abdf288d9dd8da","evidence-official-39a15ed8072aeea14a22","evidence-official-734117ed8501eb26fefc","evidence-official-7cb8cb7f091536b92c11","evidence-official-bd6ec1f819e22f075c32"],"source_locator":"src/boltz/model/models/boltz2.py: Boltz2.__init__, forward, PairformerModule, AtomDiffusion and AffinityModule; docs/prediction.md: affinity outputs; Boltz-2 report Sections 2 Data, 3 Architecture, 4 Training and 6 Limitations"},{"label":"Known versions","value":"Boltz-1 and Boltz-2 are distinct released generations; the catalogue does not select an evaluated checkpoint.","status":"source_checked","source_ids":["evidence-official-f780521ccebbb0691f9e","evidence-official-5a9e55abdf288d9dd8da","evidence-official-39a15ed8072aeea14a22","evidence-official-734117ed8501eb26fefc","evidence-official-7cb8cb7f091536b92c11","evidence-official-bd6ec1f819e22f075c32"],"source_locator":"src/boltz/model/models/boltz2.py: Boltz2.__init__, forward, PairformerModule, AtomDiffusion and AffinityModule; docs/prediction.md: affinity outputs; Boltz-2 report Sections 2 Data, 3 Architecture, 4 Training and 6 Limitations"},{"label":"Training data","value":"Boltz-2 structure training combines pre-June-2023 PDB entries, MISATO/ATLAS/mdCATH molecular dynamics, and AlphaFold2/Boltz-1 distillation. Separate affinity training uses curated PubChem, ChEMBL, BindingDB, HTS, CeMM and MIDAS evidence with different regression/classification labels.","status":"source_checked","source_ids":["evidence-official-f780521ccebbb0691f9e","evidence-official-5a9e55abdf288d9dd8da","evidence-official-39a15ed8072aeea14a22","evidence-official-734117ed8501eb26fefc","evidence-official-7cb8cb7f091536b92c11","evidence-official-bd6ec1f819e22f075c32"],"source_locator":"src/boltz/model/models/boltz2.py: Boltz2.__init__, forward, PairformerModule, AtomDiffusion and AffinityModule; docs/prediction.md: affinity outputs; Boltz-2 report Sections 2 Data, 3 Architecture, 4 Training and 6 Limitations"},{"label":"Training cutoff","value":"Boltz-2 experimental PDB structures were released before 2023-06-01. This is not a shared cutoff for every affinity, MD or distilled resource, nor a Boltz-1 training cutoff.","status":"source_checked","source_ids":["evidence-official-f780521ccebbb0691f9e","evidence-official-5a9e55abdf288d9dd8da","evidence-official-39a15ed8072aeea14a22","evidence-official-734117ed8501eb26fefc","evidence-official-7cb8cb7f091536b92c11","evidence-official-bd6ec1f819e22f075c32"],"source_locator":"src/boltz/model/models/boltz2.py: Boltz2.__init__, forward, PairformerModule, AtomDiffusion and AffinityModule; docs/prediction.md: affinity outputs; Boltz-2 report Sections 2 Data, 3 Architecture, 4 Training and 6 Limitations"},{"label":"Context limits","value":"The Boltz-2 report describes training crops up to 768 tokens. This is a training-crop size rather than a universal inference maximum; affinity additionally uses a pocket crop.","status":"source_checked","source_ids":["evidence-official-f780521ccebbb0691f9e","evidence-official-5a9e55abdf288d9dd8da","evidence-official-39a15ed8072aeea14a22","evidence-official-734117ed8501eb26fefc","evidence-official-7cb8cb7f091536b92c11","evidence-official-bd6ec1f819e22f075c32"],"source_locator":"src/boltz/model/models/boltz2.py: Boltz2.__init__, forward, PairformerModule, AtomDiffusion and AffinityModule; docs/prediction.md: affinity outputs; Boltz-2 report Sections 2 Data, 3 Architecture, 4 Training and 6 Limitations"},{"label":"Weights licence","value":"MIT; the README explicitly applies this licence to code and model weights.","status":"source_checked","source_ids":["evidence-official-f780521ccebbb0691f9e","evidence-official-5a9e55abdf288d9dd8da","evidence-official-39a15ed8072aeea14a22","evidence-official-734117ed8501eb26fefc","evidence-official-7cb8cb7f091536b92c11","evidence-official-bd6ec1f819e22f075c32"],"source_locator":"src/boltz/model/models/boltz2.py: Boltz2.__init__, forward, PairformerModule, AtomDiffusion and AffinityModule; docs/prediction.md: affinity outputs; Boltz-2 report Sections 2 Data, 3 Architecture, 4 Training and 6 Limitations"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/jwohlwend/boltz","status":"source_checked","source_ids":["evidence-official-f780521ccebbb0691f9e","evidence-official-5a9e55abdf288d9dd8da","evidence-official-39a15ed8072aeea14a22","evidence-official-734117ed8501eb26fefc","evidence-official-7cb8cb7f091536b92c11","evidence-official-bd6ec1f819e22f075c32"],"source_locator":"src/boltz/model/models/boltz2.py: Boltz2.__init__, forward, PairformerModule, AtomDiffusion and AffinityModule; docs/prediction.md: affinity outputs; Boltz-2 report Sections 2 Data, 3 Architecture, 4 Training and 6 Limitations"},{"label":"Code licence","value":"MIT","status":"source_checked","source_ids":["evidence-official-8a985eabfd054f0dec4d"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"The project distributes prediction code, model weights and training instructions.","source_ids":["evidence-official-f780521ccebbb0691f9e","evidence-official-5a9e55abdf288d9dd8da","evidence-official-39a15ed8072aeea14a22","evidence-official-734117ed8501eb26fefc","evidence-official-7cb8cb7f091536b92c11","evidence-official-bd6ec1f819e22f075c32"],"source_locator":"src/boltz/model/models/boltz2.py: Boltz2.__init__, forward, PairformerModule, AtomDiffusion and AffinityModule; docs/prediction.md: affinity outputs; Boltz-2 report Sections 2 Data, 3 Architecture, 4 Training and 6 Limitations"}],"limitations":[{"text":"Affinity predictions depend on a plausible binding pose and mix biochemical endpoint types. The report notes limited handling of cofactors, water and multimeric binding partners, and substantial variation between assays.","source_ids":["evidence-official-f780521ccebbb0691f9e","evidence-official-5a9e55abdf288d9dd8da","evidence-official-39a15ed8072aeea14a22","evidence-official-734117ed8501eb26fefc","evidence-official-7cb8cb7f091536b92c11","evidence-official-bd6ec1f819e22f075c32"],"source_locator":"src/boltz/model/models/boltz2.py: Boltz2.__init__, forward, PairformerModule, AtomDiffusion and AffinityModule; docs/prediction.md: affinity outputs; Boltz-2 report Sections 2 Data, 3 Architecture, 4 Training and 6 Limitations"}],"diagram":{"title":"Boltz-2 workflow","steps":["Molecular inputs, MSA and templates","Token and pair embeddings","Pairformer trunk","Atom-coordinate diffusion","Structure, confidence and affinity"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-f780521ccebbb0691f9e","evidence-official-5a9e55abdf288d9dd8da","evidence-official-39a15ed8072aeea14a22","evidence-official-734117ed8501eb26fefc","evidence-official-7cb8cb7f091536b92c11","evidence-official-bd6ec1f819e22f075c32"],"source_locator":"src/boltz/model/models/boltz2.py: Boltz2.__init__, forward, PairformerModule, AtomDiffusion and AffinityModule; docs/prediction.md: affinity outputs; Boltz-2 report Sections 2 Data, 3 Architecture, 4 Training and 6 Limitations"},"coverage":"limited","gaps":["Parameters: A complete parameter total is not stated in the reviewed Boltz-2 architecture report or model constructor; structure, confidence and affinity are separate modules."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}}} {"id":"catalog-model-chai-1","kind":"model","name":"Chai-1","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["proteins-complexes","molecular-interactions"],"method_types":["foundation model"]},"source_ids":["catalog-source-chai-1"],"links":[],"attributes":{"entity_level":"family","version":"released weights","reported_name":"Chai-1","access":"Public code and weights under Apache 2.0; substantial compute required.","method_type":"foundation model","historical_missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"Chai-1 predicts the structures of biomolecular complexes containing proteins, nucleic acids and small molecules.","summary_source_ids":["evidence-official-affbe2d511ff2a80f457","evidence-official-b91fc1ec601eec6598c3"],"summary_source_locator":"Chai-1 Technical Report v1 (9 September 2024), Sections 2.1, 2.7–2.8 and 4.1–4.3; current repository README Licence","sections":[{"title":"How it works","body":"Chai-1 predicts the structures of biomolecular complexes containing proteins, nucleic acids and small molecules. An AlphaFold3-like structure architecture dominated by pair-biased self-attention, with additional protein-language-model embeddings and optional inter-chain constraint features. The documented inputs are FASTA sequences, ligand SMILES and optional alignments, templates, contacts or covalent-bond restraints. The output consists of sampled complex structures; the default command produces five predictions.","source_ids":["evidence-official-affbe2d511ff2a80f457","evidence-official-b91fc1ec601eec6598c3"],"source_locator":"Chai-1 Technical Report v1 (9 September 2024), Sections 2.1, 2.7–2.8 and 4.1–4.3; current repository README Licence"},{"title":"Versions and reproducibility","body":"Chai-1; README installation example pins chai_lab 0.6.1. The applicable input limits require configuration-specific checking.","source_ids":["evidence-official-affbe2d511ff2a80f457","evidence-official-b91fc1ec601eec6598c3"],"source_locator":"Chai-1 Technical Report v1 (9 September 2024), Sections 2.1, 2.7–2.8 and 4.1–4.3; current repository README Licence"}],"facts":[{"label":"Model type","value":"Multimodal biomolecular structure predictor","status":"source_checked","source_ids":["evidence-official-affbe2d511ff2a80f457","evidence-official-b91fc1ec601eec6598c3"],"source_locator":"Chai-1 Technical Report v1 (9 September 2024), Sections 2.1, 2.7–2.8 and 4.1–4.3; current repository README Licence"},{"label":"Architecture","value":"An AlphaFold3-like structure architecture dominated by pair-biased self-attention, with additional protein-language-model embeddings and optional inter-chain constraint features.","status":"source_checked","source_ids":["evidence-official-affbe2d511ff2a80f457","evidence-official-b91fc1ec601eec6598c3"],"source_locator":"Chai-1 Technical Report v1 (9 September 2024), Sections 2.1, 2.7–2.8 and 4.1–4.3; current repository README Licence"},{"label":"Inputs","value":"FASTA sequences, ligand SMILES and optional alignments, templates, contacts or covalent-bond restraints.","status":"source_checked","source_ids":["evidence-official-affbe2d511ff2a80f457","evidence-official-b91fc1ec601eec6598c3"],"source_locator":"Chai-1 Technical Report v1 (9 September 2024), Sections 2.1, 2.7–2.8 and 4.1–4.3; current repository README Licence"},{"label":"Outputs","value":"Sampled complex structures; the default command produces five predictions.","status":"source_checked","source_ids":["evidence-official-affbe2d511ff2a80f457","evidence-official-b91fc1ec601eec6598c3"],"source_locator":"Chai-1 Technical Report v1 (9 September 2024), Sections 2.1, 2.7–2.8 and 4.1–4.3; current repository README Licence"},{"label":"Parameters","value":"The September 2024 report specifies a 3B protein-language-model component but does not state a complete predictor total in the reviewed architecture sections.","status":"unreported","source_ids":["evidence-official-affbe2d511ff2a80f457","evidence-official-b91fc1ec601eec6598c3"],"source_locator":"Chai-1 Technical Report v1 (9 September 2024), Sections 2.1, 2.7–2.8 and 4.1–4.3; current repository README Licence"},{"label":"Known versions","value":"Chai-1; README installation example pins chai_lab 0.6.1.","status":"source_checked","source_ids":["evidence-official-affbe2d511ff2a80f457","evidence-official-b91fc1ec601eec6598c3"],"source_locator":"Chai-1 Technical Report v1 (9 September 2024), Sections 2.1, 2.7–2.8 and 4.1–4.3; current repository README Licence"},{"label":"Training data","value":"PDB structures and AlphaFoldDB distillation, with optional MSAs/templates. The technical report describes no other AlphaFold3 distillation datasets; server MSA search differs from the paper evaluation pipeline.","status":"source_checked","source_ids":["evidence-official-affbe2d511ff2a80f457","evidence-official-b91fc1ec601eec6598c3"],"source_locator":"Chai-1 Technical Report v1 (9 September 2024), Sections 2.1, 2.7–2.8 and 4.1–4.3; current repository README Licence"},{"label":"Training cutoff","value":"PDB structure and PDB70-template release cutoff: 2021-01-12, according to the September 2024 technical report.","status":"source_checked","source_ids":["evidence-official-affbe2d511ff2a80f457","evidence-official-b91fc1ec601eec6598c3"],"source_locator":"Chai-1 Technical Report v1 (9 September 2024), Sections 2.1, 2.7–2.8 and 4.1–4.3; current repository README Licence"},{"label":"Context limits","value":"The reviewed report and inference README do not specify one validated maximum for all protein, nucleic-acid and ligand inputs; resource requirements and molecular composition remain relevant.","status":"unreported","source_ids":["evidence-official-affbe2d511ff2a80f457","evidence-official-b91fc1ec601eec6598c3"],"source_locator":"Chai-1 Technical Report v1 (9 September 2024), Sections 2.1, 2.7–2.8 and 4.1–4.3; current repository README Licence"},{"label":"Weights licence","value":"Apache-2.0; README Licence explicitly covers code and weights.","status":"source_checked","source_ids":["evidence-official-affbe2d511ff2a80f457","evidence-official-b91fc1ec601eec6598c3"],"source_locator":"Chai-1 Technical Report v1 (9 September 2024), Sections 2.1, 2.7–2.8 and 4.1–4.3; current repository README Licence"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/chaidiscovery/chai-lab","status":"source_checked","source_ids":["evidence-official-affbe2d511ff2a80f457","evidence-official-b91fc1ec601eec6598c3"],"source_locator":"Chai-1 Technical Report v1 (9 September 2024), Sections 2.1, 2.7–2.8 and 4.1–4.3; current repository README Licence"},{"label":"Code licence","value":"Apache-2.0","status":"source_checked","source_ids":["evidence-official-290b60bb932abe9929a2"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"Supports user-provided restraints and covalent bonds, alongside MSA and template inputs.","source_ids":["evidence-official-affbe2d511ff2a80f457","evidence-official-b91fc1ec601eec6598c3"],"source_locator":"Chai-1 Technical Report v1 (9 September 2024), Sections 2.1, 2.7–2.8 and 4.1–4.3; current repository README Licence"}],"limitations":[{"text":"The report documents failures of relative chain placement and sensitivity to modified residues. Inputs, MSA/template evidence and sampled structures must remain explicit when comparing runs.","source_ids":["evidence-official-affbe2d511ff2a80f457","evidence-official-b91fc1ec601eec6598c3"],"source_locator":"Chai-1 Technical Report v1 (9 September 2024), Sections 2.1, 2.7–2.8 and 4.1–4.3; current repository README Licence"}],"diagram":{"title":"Chai-1 workflow","steps":["Sequences and molecules","Optional MSA, template or restraints","Chai-1 inference","Complex structures"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-affbe2d511ff2a80f457","evidence-official-b91fc1ec601eec6598c3"],"source_locator":"Chai-1 Technical Report v1 (9 September 2024), Sections 2.1, 2.7–2.8 and 4.1–4.3; current repository README Licence"},"coverage":"limited","gaps":["Parameters: The September 2024 report specifies a 3B protein-language-model component but does not state a complete predictor total in the reviewed architecture sections.","Context limits: The reviewed report and inference README do not specify one validated maximum for all protein, nucleic-acid and ligand inputs; resource requirements and molecular composition remain relevant."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}}} {"id":"catalog-model-diffdock-l","kind":"model","name":"DiffDock-L","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["molecular-interactions"],"method_types":["specialist"]},"source_ids":["catalog-source-diffdock-l"],"links":[],"attributes":{"entity_level":"family","version":"2024 release","reported_name":"DiffDock-L","access":"Public pose-prediction code and weights; no native affinity prediction.","method_type":"specialist","historical_missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"DiffDock-L places small-molecule ligands in protein structures using a diffusion docking model.","summary_source_ids":["evidence-official-fa7a6c37148223466dee","evidence-official-fffd4026cb8f4018fe2e"],"summary_source_locator":"DiffDock-L paper Section 5.1 and Figure 3; official README: February 2024 update and License","sections":[{"title":"How it works","body":"DiffDock-L places small-molecule ligands in protein structures using a diffusion docking model. Diffusion-based molecular docking; the repository defaults to DiffDock-L rather than the original DiffDock model. The documented inputs are protein structure and a small-molecule ligand. The output consists of candidate ligand poses and confidence scores.","source_ids":["evidence-official-fa7a6c37148223466dee","evidence-official-fffd4026cb8f4018fe2e"],"source_locator":"DiffDock-L paper Section 5.1 and Figure 3; official README: February 2024 update and License"},{"title":"Versions and reproducibility","body":"DiffDock-L released February 2024; reproducing the original DiffDock requires its historical commit. The applicable input limits require configuration-specific checking.","source_ids":["evidence-official-fa7a6c37148223466dee","evidence-official-fffd4026cb8f4018fe2e"],"source_locator":"DiffDock-L paper Section 5.1 and Figure 3; official README: February 2024 update and License"}],"facts":[{"label":"Model type","value":"Diffusion-based molecular docking model","status":"source_checked","source_ids":["evidence-official-fa7a6c37148223466dee","evidence-official-fffd4026cb8f4018fe2e"],"source_locator":"DiffDock-L paper Section 5.1 and Figure 3; official README: February 2024 update and License"},{"label":"Architecture","value":"Diffusion-based molecular docking; the repository defaults to DiffDock-L rather than the original DiffDock model.","status":"source_checked","source_ids":["evidence-official-fa7a6c37148223466dee","evidence-official-fffd4026cb8f4018fe2e"],"source_locator":"DiffDock-L paper Section 5.1 and Figure 3; official README: February 2024 update and License"},{"label":"Inputs","value":"Protein structure and a small-molecule ligand.","status":"source_checked","source_ids":["evidence-official-fa7a6c37148223466dee","evidence-official-fffd4026cb8f4018fe2e"],"source_locator":"DiffDock-L paper Section 5.1 and Figure 3; official README: February 2024 update and License"},{"label":"Outputs","value":"Candidate ligand poses and confidence scores.","status":"source_checked","source_ids":["evidence-official-fa7a6c37148223466dee","evidence-official-fffd4026cb8f4018fe2e"],"source_locator":"DiffDock-L paper Section 5.1 and Figure 3; official README: February 2024 update and License"},{"label":"Parameters","value":"Approximately 30M in the larger score model; the confidence model is a separate component, so this is not a verified total for the complete pipeline.","status":"source_checked","source_ids":["evidence-official-fa7a6c37148223466dee","evidence-official-fffd4026cb8f4018fe2e"],"source_locator":"DiffDock-L paper Section 5.1 and Figure 3; official README: February 2024 update and License"},{"label":"Known versions","value":"DiffDock-L released February 2024; reproducing the original DiffDock requires its historical commit.","status":"source_checked","source_ids":["evidence-official-fa7a6c37148223466dee","evidence-official-fffd4026cb8f4018fe2e"],"source_locator":"DiffDock-L paper Section 5.1 and Figure 3; official README: February 2024 update and License"},{"label":"Training data","value":"PDBBind plus filtered pre-2019 Binding MOAD complexes from training/validation protein-domain clusters, with synthetic sidechain-as-ligand augmentation for additional pocket diversity.","status":"source_checked","source_ids":["evidence-official-fa7a6c37148223466dee","evidence-official-fffd4026cb8f4018fe2e"],"source_locator":"DiffDock-L paper Section 5.1 and Figure 3; official README: February 2024 update and License"},{"label":"Training cutoff","value":"The added Binding MOAD complexes were released before 2019; this is a component-specific restriction, not a universal cutoff for every input resource.","status":"source_checked","source_ids":["evidence-official-fa7a6c37148223466dee","evidence-official-fffd4026cb8f4018fe2e"],"source_locator":"DiffDock-L paper Section 5.1 and Figure 3; official README: February 2024 update and License"},{"label":"Context limits","value":"This is a protein–ligand graph model rather than a fixed text-token window. The reviewed paper and README do not establish one maximum for arbitrary receptor and ligand sizes.","status":"unreported","source_ids":["evidence-official-fa7a6c37148223466dee","evidence-official-fffd4026cb8f4018fe2e"],"source_locator":"DiffDock-L paper Section 5.1 and Figure 3; official README: February 2024 update and License"},{"label":"Weights licence","value":"MIT; README License explicitly includes code and model weights.","status":"source_checked","source_ids":["evidence-official-fa7a6c37148223466dee","evidence-official-fffd4026cb8f4018fe2e"],"source_locator":"DiffDock-L paper Section 5.1 and Figure 3; official README: February 2024 update and License"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/gcorso/DiffDock","status":"source_checked","source_ids":["evidence-official-fa7a6c37148223466dee","evidence-official-fffd4026cb8f4018fe2e"],"source_locator":"DiffDock-L paper Section 5.1 and Figure 3; official README: February 2024 update and License"},{"label":"Code licence","value":"MIT","status":"source_checked","source_ids":["evidence-official-af60e327735834e48fd7"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"The repository provides evaluation splits and instructions for PDBBind, Binding MOAD and PoseBusters.","source_ids":["evidence-official-fa7a6c37148223466dee","evidence-official-fffd4026cb8f4018fe2e"],"source_locator":"DiffDock-L paper Section 5.1 and Figure 3; official README: February 2024 update and License"}],"limitations":[{"text":"The authors restrict the intended use to small-molecule docking. Large ligands, large protein complexes and unbound receptor conformations require particular caution.","source_ids":["evidence-official-fa7a6c37148223466dee","evidence-official-fffd4026cb8f4018fe2e"],"source_locator":"DiffDock-L paper Section 5.1 and Figure 3; official README: February 2024 update and License"}],"diagram":{"title":"DiffDock-L workflow","steps":["Protein and ligand","Diffusion pose sampling","Confidence scoring","Candidate docking poses"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-fa7a6c37148223466dee","evidence-official-fffd4026cb8f4018fe2e"],"source_locator":"DiffDock-L paper Section 5.1 and Figure 3; official README: February 2024 update and License"},"coverage":"limited","gaps":["Context limits: This is a protein–ligand graph model rather than a fixed text-token window. The reviewed paper and README do not establish one maximum for arbitrary receptor and ligand sizes."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}}} {"id":"catalog-model-dnabert-2","kind":"model","name":"DNABERT-2","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["dna-genomes"],"method_types":["foundation model"]},"source_ids":["catalog-source-dnabert-2"],"links":[{"relation":"variant_of","target_id":"discovery-model-dnabert-2"}],"attributes":{"entity_level":"family","version":"117M","reported_name":"DNABERT-2","access":"Public checkpoint; remote model code needs review before local use.","method_type":"foundation model","historical_missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"DNABERT-2 learns DNA representations that can be adapted to genomic prediction tasks.","summary_source_ids":["evidence-official-7267eabca7c4a7945878","evidence-official-2cf0a41fec83ee9c5cc9","evidence-official-96b5c3a31a50f7c259d6","evidence-official-a48ae27aa000bdab7442","evidence-official-3aa749d3205c0a768b22"],"summary_source_locator":"DNABERT-2 paper Sections 3.2, 4.1 and 5.2 (Further Pre-Training); official model card and implementation README; official DNABERT-2-117M checkpoint LICENSE at revision 7bce263b15377fc15361f52cfab88f8b586abda0","sections":[{"title":"How it works","body":"DNABERT-2 merges recurring DNA substrings into byte-pair tokens, then processes those tokens with a masked-language-model transformer. ALiBi supplies distance-dependent attention biases, while FlashAttention changes how attention is computed. The resulting contextual embeddings need an explicit pooling rule and prediction head for a downstream task.","source_ids":["evidence-official-7267eabca7c4a7945878","evidence-official-2cf0a41fec83ee9c5cc9","evidence-official-96b5c3a31a50f7c259d6","evidence-official-a48ae27aa000bdab7442","evidence-official-3aa749d3205c0a768b22"],"source_locator":"DNABERT-2 paper Sections 3.2, 4.1 and 5.2 (Further Pre-Training); official model card and implementation README; official DNABERT-2-117M checkpoint LICENSE at revision 7bce263b15377fc15361f52cfab88f8b586abda0"},{"title":"Versions and reproducibility","body":"DNABERT-2-117M model card and official DNABERT_2 implementation. ALiBi permits inference beyond the pretraining sequence length, subject to attention/memory cost; this does not establish unlimited biological context or validated accuracy at arbitrary lengths.","source_ids":["evidence-official-7267eabca7c4a7945878","evidence-official-2cf0a41fec83ee9c5cc9","evidence-official-96b5c3a31a50f7c259d6","evidence-official-a48ae27aa000bdab7442","evidence-official-3aa749d3205c0a768b22"],"source_locator":"DNABERT-2 paper Sections 3.2, 4.1 and 5.2 (Further Pre-Training); official model card and implementation README; official DNABERT-2-117M checkpoint LICENSE at revision 7bce263b15377fc15361f52cfab88f8b586abda0"}],"facts":[{"label":"Model type","value":"Masked-token DNA transformer encoder","status":"source_checked","source_ids":["evidence-official-7267eabca7c4a7945878","evidence-official-2cf0a41fec83ee9c5cc9","evidence-official-96b5c3a31a50f7c259d6","evidence-official-a48ae27aa000bdab7442","evidence-official-3aa749d3205c0a768b22"],"source_locator":"DNABERT-2 paper Sections 3.2, 4.1 and 5.2 (Further Pre-Training); official model card and implementation README; official DNABERT-2-117M checkpoint LICENSE at revision 7bce263b15377fc15361f52cfab88f8b586abda0"},{"label":"Architecture","value":"BERT-style DNA encoder with byte-pair tokenization, ALiBi relative attention biases and FlashAttention; task heads and pooling are separately configured.","status":"source_checked","source_ids":["evidence-official-7267eabca7c4a7945878","evidence-official-2cf0a41fec83ee9c5cc9","evidence-official-96b5c3a31a50f7c259d6","evidence-official-a48ae27aa000bdab7442","evidence-official-3aa749d3205c0a768b22"],"source_locator":"DNABERT-2 paper Sections 3.2, 4.1 and 5.2 (Further Pre-Training); official model card and implementation README; official DNABERT-2-117M checkpoint LICENSE at revision 7bce263b15377fc15361f52cfab88f8b586abda0"},{"label":"Inputs","value":"DNA sequence tokenized with the supplied tokenizer.","status":"source_checked","source_ids":["evidence-official-7267eabca7c4a7945878","evidence-official-2cf0a41fec83ee9c5cc9","evidence-official-96b5c3a31a50f7c259d6","evidence-official-a48ae27aa000bdab7442","evidence-official-3aa749d3205c0a768b22"],"source_locator":"DNABERT-2 paper Sections 3.2, 4.1 and 5.2 (Further Pre-Training); official model card and implementation README; official DNABERT-2-117M checkpoint LICENSE at revision 7bce263b15377fc15361f52cfab88f8b586abda0"},{"label":"Outputs","value":"Token representations and, after a specified adaptation, task predictions.","status":"source_checked","source_ids":["evidence-official-7267eabca7c4a7945878","evidence-official-2cf0a41fec83ee9c5cc9","evidence-official-96b5c3a31a50f7c259d6","evidence-official-a48ae27aa000bdab7442","evidence-official-3aa749d3205c0a768b22"],"source_locator":"DNABERT-2 paper Sections 3.2, 4.1 and 5.2 (Further Pre-Training); official model card and implementation README; official DNABERT-2-117M checkpoint LICENSE at revision 7bce263b15377fc15361f52cfab88f8b586abda0"},{"label":"Parameters","value":"117 million for DNABERT-2-117M; family names do not establish a particular checkpoint.","status":"source_checked","source_ids":["evidence-official-7267eabca7c4a7945878","evidence-official-2cf0a41fec83ee9c5cc9","evidence-official-96b5c3a31a50f7c259d6","evidence-official-a48ae27aa000bdab7442","evidence-official-3aa749d3205c0a768b22"],"source_locator":"DNABERT-2 paper Sections 3.2, 4.1 and 5.2 (Further Pre-Training); official model card and implementation README; official DNABERT-2-117M checkpoint LICENSE at revision 7bce263b15377fc15361f52cfab88f8b586abda0"},{"label":"Known versions","value":"DNABERT-2-117M model card and official DNABERT_2 implementation.","status":"source_checked","source_ids":["evidence-official-7267eabca7c4a7945878","evidence-official-2cf0a41fec83ee9c5cc9","evidence-official-96b5c3a31a50f7c259d6","evidence-official-a48ae27aa000bdab7442","evidence-official-3aa749d3205c0a768b22"],"source_locator":"DNABERT-2 paper Sections 3.2, 4.1 and 5.2 (Further Pre-Training); official model card and implementation README; official DNABERT-2-117M checkpoint LICENSE at revision 7bce263b15377fc15361f52cfab88f8b586abda0"},{"label":"Training data","value":"The paper describes a 32.49-billion-base corpus covering 135 species in six groups, alongside a 2.75-billion-base human corpus. Further GUE-domain pretraining is a separately reported model variant.","status":"source_checked","source_ids":["evidence-official-7267eabca7c4a7945878","evidence-official-2cf0a41fec83ee9c5cc9","evidence-official-96b5c3a31a50f7c259d6","evidence-official-a48ae27aa000bdab7442","evidence-official-3aa749d3205c0a768b22"],"source_locator":"DNABERT-2 paper Sections 3.2, 4.1 and 5.2 (Further Pre-Training); official model card and implementation README; official DNABERT-2-117M checkpoint LICENSE at revision 7bce263b15377fc15361f52cfab88f8b586abda0"},{"label":"Training cutoff","value":"The paper identifies the human and multispecies genome corpora but does not state one latest-sequence deposition date in its reviewed pretraining-data sections.","status":"unreported","source_ids":["evidence-official-7267eabca7c4a7945878","evidence-official-2cf0a41fec83ee9c5cc9","evidence-official-96b5c3a31a50f7c259d6","evidence-official-a48ae27aa000bdab7442","evidence-official-3aa749d3205c0a768b22"],"source_locator":"DNABERT-2 paper Sections 3.2, 4.1 and 5.2 (Further Pre-Training); official model card and implementation README; official DNABERT-2-117M checkpoint LICENSE at revision 7bce263b15377fc15361f52cfab88f8b586abda0"},{"label":"Context limits","value":"ALiBi permits inference beyond the pretraining sequence length, subject to attention/memory cost; this does not establish unlimited biological context or validated accuracy at arbitrary lengths.","status":"source_checked","source_ids":["evidence-official-7267eabca7c4a7945878","evidence-official-2cf0a41fec83ee9c5cc9","evidence-official-96b5c3a31a50f7c259d6","evidence-official-a48ae27aa000bdab7442","evidence-official-3aa749d3205c0a768b22"],"source_locator":"DNABERT-2 paper Sections 3.2, 4.1 and 5.2 (Further Pre-Training); official model card and implementation README; official DNABERT-2-117M checkpoint LICENSE at revision 7bce263b15377fc15361f52cfab88f8b586abda0"},{"label":"Weights licence","value":"The official zhihan1996/DNABERT-2-117M checkpoint repository carries Apache-2.0 in its pinned LICENSE. This does not assign terms to a separately fitted downstream predictor.","status":"source_checked","source_ids":["evidence-official-7267eabca7c4a7945878","evidence-official-2cf0a41fec83ee9c5cc9","evidence-official-96b5c3a31a50f7c259d6","evidence-official-a48ae27aa000bdab7442","evidence-official-3aa749d3205c0a768b22"],"source_locator":"DNABERT-2 paper Sections 3.2, 4.1 and 5.2 (Further Pre-Training); official model card and implementation README; official DNABERT-2-117M checkpoint LICENSE at revision 7bce263b15377fc15361f52cfab88f8b586abda0"},{"label":"Access","value":"Official downloadable model/card and usage examples: https://huggingface.co/zhihan1996/DNABERT-2-117M","status":"source_checked","source_ids":["evidence-official-7267eabca7c4a7945878","evidence-official-2cf0a41fec83ee9c5cc9","evidence-official-96b5c3a31a50f7c259d6","evidence-official-a48ae27aa000bdab7442","evidence-official-3aa749d3205c0a768b22"],"source_locator":"DNABERT-2 paper Sections 3.2, 4.1 and 5.2 (Further Pre-Training); official model card and implementation README; official DNABERT-2-117M checkpoint LICENSE at revision 7bce263b15377fc15361f52cfab88f8b586abda0"},{"label":"Code licence","value":"Apache-2.0","status":"source_checked","source_ids":["evidence-official-25b222d11900e0e88a51"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"The released model supports embedding extraction and task-specific fine-tuning.","source_ids":["evidence-official-7267eabca7c4a7945878","evidence-official-2cf0a41fec83ee9c5cc9","evidence-official-96b5c3a31a50f7c259d6","evidence-official-a48ae27aa000bdab7442","evidence-official-3aa749d3205c0a768b22"],"source_locator":"DNABERT-2 paper Sections 3.2, 4.1 and 5.2 (Further Pre-Training); official model card and implementation README; official DNABERT-2-117M checkpoint LICENSE at revision 7bce263b15377fc15361f52cfab88f8b586abda0"}],"limitations":[{"text":"An embedding model alone is not the same evaluated pipeline as frozen embeddings followed by logistic regression. Tokenization and pooling choices must be preserved.","source_ids":["evidence-official-7267eabca7c4a7945878","evidence-official-2cf0a41fec83ee9c5cc9","evidence-official-96b5c3a31a50f7c259d6","evidence-official-a48ae27aa000bdab7442","evidence-official-3aa749d3205c0a768b22"],"source_locator":"DNABERT-2 paper Sections 3.2, 4.1 and 5.2 (Further Pre-Training); official model card and implementation README; official DNABERT-2-117M checkpoint LICENSE at revision 7bce263b15377fc15361f52cfab88f8b586abda0"}],"diagram":{"title":"DNABERT-2 workflow","steps":["DNA sequence","BPE tokens","Transformer encoder","Representations","Specified task head"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-7267eabca7c4a7945878","evidence-official-2cf0a41fec83ee9c5cc9","evidence-official-96b5c3a31a50f7c259d6","evidence-official-a48ae27aa000bdab7442","evidence-official-3aa749d3205c0a768b22"],"source_locator":"DNABERT-2 paper Sections 3.2, 4.1 and 5.2 (Further Pre-Training); official model card and implementation README; official DNABERT-2-117M checkpoint LICENSE at revision 7bce263b15377fc15361f52cfab88f8b586abda0"},"coverage":"limited","gaps":["Training cutoff: The paper identifies the human and multispecies genome corpora but does not state one latest-sequence deposition date in its reviewed pretraining-data sections."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}}} {"id":"catalog-model-esm-2","kind":"model","name":"ESM-2","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["proteins-complexes"],"method_types":["foundation model"]},"source_ids":["catalog-source-esm-2"],"links":[{"relation":"variant_of","target_id":"discovery-model-esm-2"}],"attributes":{"entity_level":"family","version":"8M","reported_name":"ESM-2","access":"Public checkpoint; small 8M variant suits a local pilot.","method_type":"foundation model","historical_missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"ESM-2 is a family of protein sequence encoders that produce representations for downstream protein analyses.","summary_source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-d7e5c63feb0e0e627625","evidence-official-28fc17fbe0c1d99da219"],"summary_source_locator":"ESM README: Pre-trained Models and Main models; official facebook/esm2_t33_650M_UR50D README licence metadata and config.json","sections":[{"title":"How it works","body":"ESM-2 tokenizes an amino-acid sequence and uses a transformer encoder trained to recover masked residues. Self-attention lets each residue representation depend on its sequence context. The released model returns token probabilities and embeddings; a specified pooling rule, task head or complete folding pipeline is needed for a particular biological prediction.","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-d7e5c63feb0e0e627625","evidence-official-28fc17fbe0c1d99da219"],"source_locator":"ESM README: Pre-trained Models and Main models; official facebook/esm2_t33_650M_UR50D README licence metadata and config.json"},{"title":"Versions and reproducibility","body":"ESM-2 checkpoint identifiers encode layer count, parameter scale and training-data tag. The checked esm2_t33_650M_UR50D configuration lists max_position_embeddings=1,026. This configuration field includes model positions and is not a claim of training or validated inference on 1,026 amino acids.","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-d7e5c63feb0e0e627625","evidence-official-28fc17fbe0c1d99da219"],"source_locator":"ESM README: Pre-trained Models and Main models; official facebook/esm2_t33_650M_UR50D README licence metadata and config.json"}],"facts":[{"label":"Model type","value":"Masked-token protein transformer encoder","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-d7e5c63feb0e0e627625","evidence-official-28fc17fbe0c1d99da219"],"source_locator":"ESM README: Pre-trained Models and Main models; official facebook/esm2_t33_650M_UR50D README licence metadata and config.json"},{"label":"Architecture","value":"Masked-token protein transformer encoder; the checked 650M checkpoint has 33 layers, hidden width 1,280, 20 attention heads and rotary positional encoding.","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-d7e5c63feb0e0e627625","evidence-official-28fc17fbe0c1d99da219"],"source_locator":"ESM README: Pre-trained Models and Main models; official facebook/esm2_t33_650M_UR50D README licence metadata and config.json"},{"label":"Inputs","value":"Single amino-acid sequences.","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-d7e5c63feb0e0e627625","evidence-official-28fc17fbe0c1d99da219"],"source_locator":"ESM README: Pre-trained Models and Main models; official facebook/esm2_t33_650M_UR50D README licence metadata and config.json"},{"label":"Outputs","value":"Residue embeddings, sequence representations and masked-token predictions.","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-d7e5c63feb0e0e627625","evidence-official-28fc17fbe0c1d99da219"],"source_locator":"ESM README: Pre-trained Models and Main models; official facebook/esm2_t33_650M_UR50D README licence metadata and config.json"},{"label":"Parameters","value":"Released scales: 8M, 35M, 150M, 650M, 3B and 15B.","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-d7e5c63feb0e0e627625","evidence-official-28fc17fbe0c1d99da219"],"source_locator":"ESM README: Pre-trained Models and Main models; official facebook/esm2_t33_650M_UR50D README licence metadata and config.json"},{"label":"Known versions","value":"ESM-2 checkpoint identifiers encode layer count, parameter scale and training-data tag.","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-d7e5c63feb0e0e627625","evidence-official-28fc17fbe0c1d99da219"],"source_locator":"ESM README: Pre-trained Models and Main models; official facebook/esm2_t33_650M_UR50D README licence metadata and config.json"},{"label":"Training data","value":"UniRef50 clusters with UniRef90 sampling; the pretrained-model table labels UR50/D 2021_04.","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-d7e5c63feb0e0e627625","evidence-official-28fc17fbe0c1d99da219"],"source_locator":"ESM README: Pre-trained Models and Main models; official facebook/esm2_t33_650M_UR50D README licence metadata and config.json"},{"label":"Training cutoff","value":"The pretrained-model table identifies training-data release UR50/D 2021_04; a corpus release date is not necessarily a last-deposited-sequence cutoff.","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-d7e5c63feb0e0e627625","evidence-official-28fc17fbe0c1d99da219"],"source_locator":"ESM README: Pre-trained Models and Main models; official facebook/esm2_t33_650M_UR50D README licence metadata and config.json"},{"label":"Context limits","value":"The checked esm2_t33_650M_UR50D configuration lists max_position_embeddings=1,026. This configuration field includes model positions and is not a claim of training or validated inference on 1,026 amino acids.","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-d7e5c63feb0e0e627625","evidence-official-28fc17fbe0c1d99da219"],"source_locator":"ESM README: Pre-trained Models and Main models; official facebook/esm2_t33_650M_UR50D README licence metadata and config.json"},{"label":"Weights licence","value":"The official facebook/esm2_t33_650M_UR50D model card declares MIT; this is the inspected checkpoint, not a licence inference from source code.","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-d7e5c63feb0e0e627625","evidence-official-28fc17fbe0c1d99da219"],"source_locator":"ESM README: Pre-trained Models and Main models; official facebook/esm2_t33_650M_UR50D README licence metadata and config.json"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/facebookresearch/esm","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-d7e5c63feb0e0e627625","evidence-official-28fc17fbe0c1d99da219"],"source_locator":"ESM README: Pre-trained Models and Main models; official facebook/esm2_t33_650M_UR50D README licence metadata and config.json"},{"label":"Code licence","value":"MIT","status":"source_checked","source_ids":["evidence-official-2e7c7649620407f50f6b"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"Released checkpoints span several sizes and can be used without constructing a multiple sequence alignment.","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-d7e5c63feb0e0e627625","evidence-official-28fc17fbe0c1d99da219"],"source_locator":"ESM README: Pre-trained Models and Main models; official facebook/esm2_t33_650M_UR50D README licence metadata and config.json"}],"limitations":[{"text":"A general embedding is not a directly measured function or structure. Fine-tuning, pooling and downstream heads remain part of each evaluated configuration.","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-d7e5c63feb0e0e627625","evidence-official-28fc17fbe0c1d99da219"],"source_locator":"ESM README: Pre-trained Models and Main models; official facebook/esm2_t33_650M_UR50D README licence metadata and config.json"}],"diagram":{"title":"ESM-2 workflow","steps":["Protein sequence","Transformer layers","Residue embeddings","Specified downstream analysis"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-d7e5c63feb0e0e627625","evidence-official-28fc17fbe0c1d99da219"],"source_locator":"ESM README: Pre-trained Models and Main models; official facebook/esm2_t33_650M_UR50D README licence metadata and config.json"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}}} {"id":"catalog-model-esmfold","kind":"model","name":"ESMFold","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["proteins-complexes"],"method_types":["foundation model"]},"source_ids":["catalog-source-esmfold"],"links":[{"relation":"variant_of","target_id":"discovery-model-esmfold"}],"attributes":{"entity_level":"family","version":"v1","reported_name":"ESMFold","access":"Public checkpoint; materially larger than ESM-2 8M.","method_type":"foundation model","historical_missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"ESMFold predicts protein structures directly from amino-acid sequence using ESM-2 representations.","summary_source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-fa53ab621979fb5f37c8","evidence-official-867f2d1dcb4084248daa"],"summary_source_locator":"ESM README: ESMFold Structure Prediction; official facebook/esmfold_v1 model card and config.json esmfold_config","sections":[{"title":"How it works","body":"ESMFold predicts protein structures directly from amino-acid sequence using ESM-2 representations. ESM-2 sequence representations feed a folding trunk and structure module. The checked v1 configuration has 48 trunk blocks, eight structure-module blocks and up to four recycles. The documented inputs are protein amino-acid sequence; the ESMFold interface also accepts chains separated by a colon. The output consists of predicted PDB structure and confidence values.","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-fa53ab621979fb5f37c8","evidence-official-867f2d1dcb4084248daa"],"source_locator":"ESM README: ESMFold Structure Prediction; official facebook/esmfold_v1 model card and config.json esmfold_config"},{"title":"Versions and reproducibility","body":"esmfold_v0 and esmfold_v1; v1 is the repository recommendation. Inference length is constrained by memory; the repository documents chunking and CPU offload. Backbone position settings do not alone establish the full folding pipeline limit.","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-fa53ab621979fb5f37c8","evidence-official-867f2d1dcb4084248daa"],"source_locator":"ESM README: ESMFold Structure Prediction; official facebook/esmfold_v1 model card and config.json esmfold_config"}],"facts":[{"label":"Model type","value":"Sequence-to-structure protein prediction pipeline","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-fa53ab621979fb5f37c8","evidence-official-867f2d1dcb4084248daa"],"source_locator":"ESM README: ESMFold Structure Prediction; official facebook/esmfold_v1 model card and config.json esmfold_config"},{"label":"Architecture","value":"ESM-2 sequence representations feed a folding trunk and structure module. The checked v1 configuration has 48 trunk blocks, eight structure-module blocks and up to four recycles.","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-fa53ab621979fb5f37c8","evidence-official-867f2d1dcb4084248daa"],"source_locator":"ESM README: ESMFold Structure Prediction; official facebook/esmfold_v1 model card and config.json esmfold_config"},{"label":"Inputs","value":"Protein amino-acid sequence; the ESMFold interface also accepts chains separated by a colon.","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-fa53ab621979fb5f37c8","evidence-official-867f2d1dcb4084248daa"],"source_locator":"ESM README: ESMFold Structure Prediction; official facebook/esmfold_v1 model card and config.json esmfold_config"},{"label":"Outputs","value":"Predicted PDB structure and confidence values.","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-fa53ab621979fb5f37c8","evidence-official-867f2d1dcb4084248daa"],"source_locator":"ESM README: ESMFold Structure Prediction; official facebook/esmfold_v1 model card and config.json esmfold_config"},{"label":"Parameters","value":"The released v1 configuration identifies an ESM-2 3B backbone plus a folding trunk and structure module. The 3B figure is not the total size of the complete predictor.","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-fa53ab621979fb5f37c8","evidence-official-867f2d1dcb4084248daa"],"source_locator":"ESM README: ESMFold Structure Prediction; official facebook/esmfold_v1 model card and config.json esmfold_config"},{"label":"Known versions","value":"esmfold_v0 and esmfold_v1; v1 is the repository recommendation.","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-fa53ab621979fb5f37c8","evidence-official-867f2d1dcb4084248daa"],"source_locator":"ESM README: ESMFold Structure Prediction; official facebook/esmfold_v1 model card and config.json esmfold_config"},{"label":"Training data","value":"PDB and UniRef50 are listed for ESMFold. Full structural training-cutoff verification remains outstanding.","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-fa53ab621979fb5f37c8","evidence-official-867f2d1dcb4084248daa"],"source_locator":"ESM README: ESMFold Structure Prediction; official facebook/esmfold_v1 model card and config.json esmfold_config"},{"label":"Training cutoff","value":"PDB and UniRef50 are identified in the official model table; the inspected ESMFold-v1 card and configuration do not supply a shared latest-data date for both components.","status":"unreported","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-fa53ab621979fb5f37c8","evidence-official-867f2d1dcb4084248daa"],"source_locator":"ESM README: ESMFold Structure Prediction; official facebook/esmfold_v1 model card and config.json esmfold_config"},{"label":"Context limits","value":"Inference length is constrained by memory; the repository documents chunking and CPU offload. Backbone position settings do not alone establish the full folding pipeline limit.","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-fa53ab621979fb5f37c8","evidence-official-867f2d1dcb4084248daa"],"source_locator":"ESM README: ESMFold Structure Prediction; official facebook/esmfold_v1 model card and config.json esmfold_config"},{"label":"Weights licence","value":"MIT declared by the official facebook/esmfold_v1 model card.","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-fa53ab621979fb5f37c8","evidence-official-867f2d1dcb4084248daa"],"source_locator":"ESM README: ESMFold Structure Prediction; official facebook/esmfold_v1 model card and config.json esmfold_config"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/facebookresearch/esm","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-fa53ab621979fb5f37c8","evidence-official-867f2d1dcb4084248daa"],"source_locator":"ESM README: ESMFold Structure Prediction; official facebook/esmfold_v1 model card and config.json esmfold_config"},{"label":"Code licence","value":"MIT","status":"source_checked","source_ids":["evidence-official-2e7c7649620407f50f6b"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"Supports sequence-only structure prediction without an MSA search.","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-fa53ab621979fb5f37c8","evidence-official-867f2d1dcb4084248daa"],"source_locator":"ESM README: ESMFold Structure Prediction; official facebook/esmfold_v1 model card and config.json esmfold_config"}],"limitations":[{"text":"ESMFold v0 and v1 are different releases. The repository discourages using structure-module-only ablation models as the standard predictor.","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-fa53ab621979fb5f37c8","evidence-official-867f2d1dcb4084248daa"],"source_locator":"ESM README: ESMFold Structure Prediction; official facebook/esmfold_v1 model card and config.json esmfold_config"}],"diagram":{"title":"ESMFold workflow","steps":["Protein sequence","ESM-2 representations","Folding module","Predicted structure and confidence"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-fa53ab621979fb5f37c8","evidence-official-867f2d1dcb4084248daa"],"source_locator":"ESM README: ESMFold Structure Prediction; official facebook/esmfold_v1 model card and config.json esmfold_config"},"coverage":"limited","gaps":["Training cutoff: PDB and UniRef50 are identified in the official model table; the inspected ESMFold-v1 card and configuration do not supply a shared latest-data date for both components."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}}} {"id":"catalog-model-evo-2","kind":"model","name":"Evo 2","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["dna-genomes","microbes-communities"],"method_types":["foundation model"]},"source_ids":["catalog-source-evo-2"],"links":[],"attributes":{"entity_level":"family","version":"7B","reported_name":"Evo 2","access":"Public checkpoints; official local inference needs CUDA hardware and substantial memory.","method_type":"foundation model","historical_missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"Evo 2 models and generates DNA over long contexts at single-nucleotide resolution.","summary_source_ids":["evidence-official-a248990b44d481814f47","evidence-official-849b751ed4bf88622749","evidence-official-b8df803e8f55324a4eb5","evidence-official-8717ffb993cc8f7eddec"],"summary_source_locator":"Evo 2 paper: architecture, training, and data, Figure 1 and Discussion; official checkpoints table and evo2_7b licence metadata","sections":[{"title":"How it works","body":"Evo 2 models and generates DNA over long contexts at single-nucleotide resolution. StripedHyena 2 hybrid architecture combining short, medium and long convolution operators with attention, trained autoregressively at single-base resolution. The documented inputs are DNA sequences represented at single-base resolution. The output consists of next-token outputs, embeddings and generated DNA sequences.","source_ids":["evidence-official-a248990b44d481814f47","evidence-official-849b751ed4bf88622749","evidence-official-b8df803e8f55324a4eb5","evidence-official-8717ffb993cc8f7eddec"],"source_locator":"Evo 2 paper: architecture, training, and data, Figure 1 and Discussion; official checkpoints table and evo2_7b licence metadata"},{"title":"Versions and reproducibility","body":"Base 8K models, long-context 1M models, 7B 262K model and separately fine-tuned Microviridae model. Checkpoint-dependent: 8K, 262K or 1M bases as listed in the Checkpoints table.","source_ids":["evidence-official-a248990b44d481814f47","evidence-official-849b751ed4bf88622749","evidence-official-b8df803e8f55324a4eb5","evidence-official-8717ffb993cc8f7eddec"],"source_locator":"Evo 2 paper: architecture, training, and data, Figure 1 and Discussion; official checkpoints table and evo2_7b licence metadata"}],"facts":[{"label":"Model type","value":"Autoregressive DNA model with StripedHyena 2","status":"source_checked","source_ids":["evidence-official-a248990b44d481814f47","evidence-official-849b751ed4bf88622749","evidence-official-b8df803e8f55324a4eb5","evidence-official-8717ffb993cc8f7eddec"],"source_locator":"Evo 2 paper: architecture, training, and data, Figure 1 and Discussion; official checkpoints table and evo2_7b licence metadata"},{"label":"Architecture","value":"StripedHyena 2 hybrid architecture combining short, medium and long convolution operators with attention, trained autoregressively at single-base resolution.","status":"source_checked","source_ids":["evidence-official-a248990b44d481814f47","evidence-official-849b751ed4bf88622749","evidence-official-b8df803e8f55324a4eb5","evidence-official-8717ffb993cc8f7eddec"],"source_locator":"Evo 2 paper: architecture, training, and data, Figure 1 and Discussion; official checkpoints table and evo2_7b licence metadata"},{"label":"Inputs","value":"DNA sequences represented at single-base resolution.","status":"source_checked","source_ids":["evidence-official-a248990b44d481814f47","evidence-official-849b751ed4bf88622749","evidence-official-b8df803e8f55324a4eb5","evidence-official-8717ffb993cc8f7eddec"],"source_locator":"Evo 2 paper: architecture, training, and data, Figure 1 and Discussion; official checkpoints table and evo2_7b licence metadata"},{"label":"Outputs","value":"Next-token outputs, embeddings and generated DNA sequences.","status":"source_checked","source_ids":["evidence-official-a248990b44d481814f47","evidence-official-849b751ed4bf88622749","evidence-official-b8df803e8f55324a4eb5","evidence-official-8717ffb993cc8f7eddec"],"source_locator":"Evo 2 paper: architecture, training, and data, Figure 1 and Discussion; official checkpoints table and evo2_7b licence metadata"},{"label":"Parameters","value":"1B, 7B, 20B and 40B checkpoints are listed.","status":"source_checked","source_ids":["evidence-official-a248990b44d481814f47","evidence-official-849b751ed4bf88622749","evidence-official-b8df803e8f55324a4eb5","evidence-official-8717ffb993cc8f7eddec"],"source_locator":"Evo 2 paper: architecture, training, and data, Figure 1 and Discussion; official checkpoints table and evo2_7b licence metadata"},{"label":"Known versions","value":"Base 8K models, long-context 1M models, 7B 262K model and separately fine-tuned Microviridae model.","status":"source_checked","source_ids":["evidence-official-a248990b44d481814f47","evidence-official-849b751ed4bf88622749","evidence-official-b8df803e8f55324a4eb5","evidence-official-8717ffb993cc8f7eddec"],"source_locator":"Evo 2 paper: architecture, training, and data, Figure 1 and Discussion; official checkpoints table and evo2_7b licence metadata"},{"label":"Training data","value":"OpenGenome2 contains more than 8.8T curated nucleotides across bacteria, archaea, eukaryotes and bacteriophage. The paper separates 2.4T tokens of training exposure for 7B from 9.3T for 40B; eukaryotic-host viral sequences were excluded.","status":"source_checked","source_ids":["evidence-official-a248990b44d481814f47","evidence-official-849b751ed4bf88622749","evidence-official-b8df803e8f55324a4eb5","evidence-official-8717ffb993cc8f7eddec"],"source_locator":"Evo 2 paper: architecture, training, and data, Figure 1 and Discussion; official checkpoints table and evo2_7b licence metadata"},{"label":"Training cutoff","value":"OpenGenome2 combines multiple nucleotide collections. The inspected paper and released checkpoint documentation do not provide one latest-deposition date that covers every component.","status":"unreported","source_ids":["evidence-official-a248990b44d481814f47","evidence-official-849b751ed4bf88622749","evidence-official-b8df803e8f55324a4eb5","evidence-official-8717ffb993cc8f7eddec"],"source_locator":"Evo 2 paper: architecture, training, and data, Figure 1 and Discussion; official checkpoints table and evo2_7b licence metadata"},{"label":"Context limits","value":"Checkpoint-dependent: 8K, 262K or 1M bases as listed in the Checkpoints table.","status":"source_checked","source_ids":["evidence-official-a248990b44d481814f47","evidence-official-849b751ed4bf88622749","evidence-official-b8df803e8f55324a4eb5","evidence-official-8717ffb993cc8f7eddec"],"source_locator":"Evo 2 paper: architecture, training, and data, Figure 1 and Discussion; official checkpoints table and evo2_7b licence metadata"},{"label":"Weights licence","value":"Apache-2.0 is declared in the inspected ArcInstitute/evo2_7b model card; other checkpoints require their own pinned terms.","status":"source_checked","source_ids":["evidence-official-a248990b44d481814f47","evidence-official-849b751ed4bf88622749","evidence-official-b8df803e8f55324a4eb5","evidence-official-8717ffb993cc8f7eddec"],"source_locator":"Evo 2 paper: architecture, training, and data, Figure 1 and Discussion; official checkpoints table and evo2_7b licence metadata"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/ArcInstitute/evo2","status":"source_checked","source_ids":["evidence-official-a248990b44d481814f47","evidence-official-849b751ed4bf88622749","evidence-official-b8df803e8f55324a4eb5","evidence-official-8717ffb993cc8f7eddec"],"source_locator":"Evo 2 paper: architecture, training, and data, Figure 1 and Discussion; official checkpoints table and evo2_7b licence metadata"},{"label":"Code licence","value":"Apache-2.0","status":"source_checked","source_ids":["evidence-official-67ee1cc31060ba8c9569"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"Different released context lengths and model scales support a range of sequence modeling workflows.","source_ids":["evidence-official-a248990b44d481814f47","evidence-official-849b751ed4bf88622749","evidence-official-b8df803e8f55324a4eb5","evidence-official-8717ffb993cc8f7eddec"],"source_locator":"Evo 2 paper: architecture, training, and data, Figure 1 and Discussion; official checkpoints table and evo2_7b licence metadata"}],"limitations":[{"text":"Hardware requirements differ by checkpoint: the README requires FP8/Transformer Engine and Hopper GPUs for some scales, while 7B supports bfloat 16 on a wider set of GPUs.","source_ids":["evidence-official-a248990b44d481814f47","evidence-official-849b751ed4bf88622749","evidence-official-b8df803e8f55324a4eb5","evidence-official-8717ffb993cc8f7eddec"],"source_locator":"Evo 2 paper: architecture, training, and data, Figure 1 and Discussion; official checkpoints table and evo2_7b licence metadata"}],"diagram":{"title":"Evo 2 workflow","steps":["DNA bases","StripedHyena 2","Autoregressive outputs","Sequence scoring or generation"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-a248990b44d481814f47","evidence-official-849b751ed4bf88622749","evidence-official-b8df803e8f55324a4eb5","evidence-official-8717ffb993cc8f7eddec"],"source_locator":"Evo 2 paper: architecture, training, and data, Figure 1 and Discussion; official checkpoints table and evo2_7b licence metadata"},"coverage":"limited","gaps":["Training cutoff: OpenGenome2 combines multiple nucleotide collections. The inspected paper and released checkpoint documentation do not provide one latest-deposition date that covers every component."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}}} {"id":"catalog-model-gears","kind":"model","name":"GEARS","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["cells-tissues"],"method_types":["specialist"]},"source_ids":["catalog-source-gears"],"links":[{"relation":"family","target_id":"discovery-model-gears"}],"attributes":{"entity_level":"family","version":"published implementation","reported_name":"GEARS","access":"Public code; task-specific training data required.","method_type":"specialist","historical_missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"GEARS predicts transcriptional responses to single- and multi-gene perturbations from single-cell perturbation screens.","summary_source_ids":["evidence-official-c037e3419c04936262a0","evidence-official-b5e81b2c497214b18f8a","evidence-official-f464cd407a6a9d031720"],"summary_source_locator":"GEARS paper Methods: Overview, Gene coexpression graph encoder, GO graph, Cross-gene effects and gene-specific decoder; gears/model.py: GEARS_Model","sections":[{"title":"How it works","body":"GEARS predicts transcriptional responses to single- and multi-gene perturbations from single-cell perturbation screens. Two graph encoders represent gene coexpression and Gene Ontology perturbation similarity. Perturbation embeddings are composed with gene embeddings, then a cross-gene network and gene-specific decoders predict expression changes. The documented inputs are single-cell expression data, perturbation labels and the graph resources used by the configured model. The output consists of predicted gene-expression responses and genetic-interaction analyses.","source_ids":["evidence-official-c037e3419c04936262a0","evidence-official-b5e81b2c497214b18f8a","evidence-official-f464cd407a6a9d031720"],"source_locator":"GEARS paper Methods: Overview, Gene coexpression graph encoder, GO graph, Cross-gene effects and gene-specific decoder; gears/model.py: GEARS_Model"},{"title":"Versions and reproducibility","body":"The README describes v0.1.1 updates; a specific trained checkpoint must be recorded separately. A gene-expression vector and perturbation set over the configured gene inventory; no fixed nucleotide or amino-acid token window.","source_ids":["evidence-official-c037e3419c04936262a0","evidence-official-b5e81b2c497214b18f8a","evidence-official-f464cd407a6a9d031720"],"source_locator":"GEARS paper Methods: Overview, Gene coexpression graph encoder, GO graph, Cross-gene effects and gene-specific decoder; gears/model.py: GEARS_Model"}],"facts":[{"label":"Model type","value":"Graph-based perturbation-response predictor","status":"source_checked","source_ids":["evidence-official-c037e3419c04936262a0","evidence-official-b5e81b2c497214b18f8a","evidence-official-f464cd407a6a9d031720"],"source_locator":"GEARS paper Methods: Overview, Gene coexpression graph encoder, GO graph, Cross-gene effects and gene-specific decoder; gears/model.py: GEARS_Model"},{"label":"Architecture","value":"Two graph encoders represent gene coexpression and Gene Ontology perturbation similarity. Perturbation embeddings are composed with gene embeddings, then a cross-gene network and gene-specific decoders predict expression changes.","status":"source_checked","source_ids":["evidence-official-c037e3419c04936262a0","evidence-official-b5e81b2c497214b18f8a","evidence-official-f464cd407a6a9d031720"],"source_locator":"GEARS paper Methods: Overview, Gene coexpression graph encoder, GO graph, Cross-gene effects and gene-specific decoder; gears/model.py: GEARS_Model"},{"label":"Inputs","value":"Single-cell expression data, perturbation labels and the graph resources used by the configured model.","status":"source_checked","source_ids":["evidence-official-c037e3419c04936262a0","evidence-official-b5e81b2c497214b18f8a","evidence-official-f464cd407a6a9d031720"],"source_locator":"GEARS paper Methods: Overview, Gene coexpression graph encoder, GO graph, Cross-gene effects and gene-specific decoder; gears/model.py: GEARS_Model"},{"label":"Outputs","value":"Predicted gene-expression responses and genetic-interaction analyses.","status":"source_checked","source_ids":["evidence-official-c037e3419c04936262a0","evidence-official-b5e81b2c497214b18f8a","evidence-official-f464cd407a6a9d031720"],"source_locator":"GEARS paper Methods: Overview, Gene coexpression graph encoder, GO graph, Cross-gene effects and gene-specific decoder; gears/model.py: GEARS_Model"},{"label":"Parameters","value":"Configuration-dependent: gene and perturbation embedding tables grow with the selected gene/perturbation inventory, alongside graph and decoder parameters.","status":"source_checked","source_ids":["evidence-official-c037e3419c04936262a0","evidence-official-b5e81b2c497214b18f8a","evidence-official-f464cd407a6a9d031720"],"source_locator":"GEARS paper Methods: Overview, Gene coexpression graph encoder, GO graph, Cross-gene effects and gene-specific decoder; gears/model.py: GEARS_Model"},{"label":"Known versions","value":"The README describes v0.1.1 updates; a specific trained checkpoint must be recorded separately.","status":"source_checked","source_ids":["evidence-official-c037e3419c04936262a0","evidence-official-b5e81b2c497214b18f8a","evidence-official-f464cd407a6a9d031720"],"source_locator":"GEARS paper Methods: Overview, Gene coexpression graph encoder, GO graph, Cross-gene effects and gene-specific decoder; gears/model.py: GEARS_Model"},{"label":"Training data","value":"Fitted to the selected perturbation screen. Examples include Norman, Adamson and Dixit; the repository also lists Replogle RPE1/K562 loaders.","status":"source_checked","source_ids":["evidence-official-c037e3419c04936262a0","evidence-official-b5e81b2c497214b18f8a","evidence-official-f464cd407a6a9d031720"],"source_locator":"GEARS paper Methods: Overview, Gene coexpression graph encoder, GO graph, Cross-gene effects and gene-specific decoder; gears/model.py: GEARS_Model"},{"label":"Training cutoff","value":"Inapplicable as a universal pretrained-model cutoff: GEARS fits the provided perturbation training set and builds its coexpression graph from that set.","status":"inapplicable","source_ids":["evidence-official-c037e3419c04936262a0","evidence-official-b5e81b2c497214b18f8a","evidence-official-f464cd407a6a9d031720"],"source_locator":"GEARS paper Methods: Overview, Gene coexpression graph encoder, GO graph, Cross-gene effects and gene-specific decoder; gears/model.py: GEARS_Model"},{"label":"Context limits","value":"A gene-expression vector and perturbation set over the configured gene inventory; no fixed nucleotide or amino-acid token window.","status":"source_checked","source_ids":["evidence-official-c037e3419c04936262a0","evidence-official-b5e81b2c497214b18f8a","evidence-official-f464cd407a6a9d031720"],"source_locator":"GEARS paper Methods: Overview, Gene coexpression graph encoder, GO graph, Cross-gene effects and gene-specific decoder; gears/model.py: GEARS_Model"},{"label":"Weights licence","value":"Separate checkpoint-distribution terms are not stated in the inspected release documentation and licence material. The source-code licence alone is not recorded as an explicit weight grant.","status":"unreported","source_ids":["evidence-official-c037e3419c04936262a0","evidence-official-b5e81b2c497214b18f8a","evidence-official-f464cd407a6a9d031720","evidence-official-f06a8695b2915f86a45a"],"source_locator":"GEARS paper Methods: Overview, Gene coexpression graph encoder, GO graph, Cross-gene effects and gene-specific decoder; gears/model.py: GEARS_Model; LICENSE: licence text"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/snap-stanford/GEARS","status":"source_checked","source_ids":["evidence-official-c037e3419c04936262a0","evidence-official-b5e81b2c497214b18f8a","evidence-official-f464cd407a6a9d031720"],"source_locator":"GEARS paper Methods: Overview, Gene coexpression graph encoder, GO graph, Cross-gene effects and gene-specific decoder; gears/model.py: GEARS_Model"},{"label":"Code licence","value":"MIT","status":"source_checked","source_ids":["evidence-official-f06a8695b2915f86a45a"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"Provides documented dataset/split handling and training interfaces for single and combinatorial perturbations.","source_ids":["evidence-official-c037e3419c04936262a0","evidence-official-b5e81b2c497214b18f8a","evidence-official-f464cd407a6a9d031720"],"source_locator":"GEARS paper Methods: Overview, Gene coexpression graph encoder, GO graph, Cross-gene effects and gene-specific decoder; gears/model.py: GEARS_Model"}],"limitations":[{"text":"The authors explicitly warn against cross-cell-type transfer, bulk-RNA assumptions and predicting combinations after training only on single perturbations.","source_ids":["evidence-official-c037e3419c04936262a0","evidence-official-b5e81b2c497214b18f8a","evidence-official-f464cd407a6a9d031720"],"source_locator":"GEARS paper Methods: Overview, Gene coexpression graph encoder, GO graph, Cross-gene effects and gene-specific decoder; gears/model.py: GEARS_Model"}],"diagram":{"title":"GEARS workflow","steps":["Perturbation screen","Gene and perturbation graph representations","GEARS prediction","Expression response"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-c037e3419c04936262a0","evidence-official-b5e81b2c497214b18f8a","evidence-official-f464cd407a6a9d031720"],"source_locator":"GEARS paper Methods: Overview, Gene coexpression graph encoder, GO graph, Cross-gene effects and gene-specific decoder; gears/model.py: GEARS_Model"},"coverage":"limited","gaps":["Weights licence: Separate checkpoint-distribution terms are not stated in the inspected release documentation and licence material. The source-code licence alone is not recorded as an explicit weight grant."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}}} {"id":"catalog-model-geneformer","kind":"model","name":"Geneformer","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["cells-tissues"],"method_types":["foundation model"]},"source_ids":["catalog-source-geneformer"],"links":[],"attributes":{"entity_level":"family","version":"published checkpoints","reported_name":"Geneformer","access":"Public checkpoints; specify exact version before evaluation.","method_type":"foundation model","historical_missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"Geneformer represents single-cell transcriptomes as ranked genes and learns contextual gene and cell representations.","summary_source_ids":["evidence-official-d3d9afa5c6172982ed01","evidence-official-f60303c5eaa60ed9c77a"],"summary_source_locator":"README.md: Model Description, pretrained model list, Application, Installation and licence frontmatter","sections":[{"title":"How it works","body":"Geneformer represents single-cell transcriptomes as ranked genes and learns contextual gene and cell representations. Transformer encoder trained to recover masked genes from rank-value-encoded expression profiles. The documented inputs are single-cell gene expression converted to corpus-normalized gene ranks. The output consists of contextual gene and cell representations; task-specific outputs after the documented fine-tuning or perturbation workflow.","source_ids":["evidence-official-d3d9afa5c6172982ed01","evidence-official-f60303c5eaa60ed9c77a"],"source_locator":"README.md: Model Description, pretrained model list, Application, Installation and licence frontmatter"},{"title":"Versions and reproducibility","body":"Geneformer-V1-10M, V2-104M, V2-316M and V2-104M_CLcancer; the card states V2-316M is the repository default. V1: 2,048 gene tokens; V2: 4,096. Vocabularies also differ.","source_ids":["evidence-official-d3d9afa5c6172982ed01","evidence-official-f60303c5eaa60ed9c77a"],"source_locator":"README.md: Model Description, pretrained model list, Application, Installation and licence frontmatter"}],"facts":[{"label":"Model type","value":"Masked-gene transcriptomic transformer encoder","status":"source_checked","source_ids":["evidence-official-d3d9afa5c6172982ed01","evidence-official-f60303c5eaa60ed9c77a"],"source_locator":"README.md: Model Description, pretrained model list, Application, Installation and licence frontmatter"},{"label":"Architecture","value":"Transformer encoder trained to recover masked genes from rank-value-encoded expression profiles.","status":"source_checked","source_ids":["evidence-official-d3d9afa5c6172982ed01","evidence-official-f60303c5eaa60ed9c77a"],"source_locator":"README.md: Model Description, pretrained model list, Application, Installation and licence frontmatter"},{"label":"Inputs","value":"Single-cell gene expression converted to corpus-normalized gene ranks.","status":"source_checked","source_ids":["evidence-official-d3d9afa5c6172982ed01","evidence-official-f60303c5eaa60ed9c77a"],"source_locator":"README.md: Model Description, pretrained model list, Application, Installation and licence frontmatter"},{"label":"Outputs","value":"Contextual gene and cell representations; task-specific outputs after the documented fine-tuning or perturbation workflow.","status":"source_checked","source_ids":["evidence-official-d3d9afa5c6172982ed01","evidence-official-f60303c5eaa60ed9c77a"],"source_locator":"README.md: Model Description, pretrained model list, Application, Installation and licence frontmatter"},{"label":"Parameters","value":"V1: 10M; V2: 104M or 316M.","status":"source_checked","source_ids":["evidence-official-d3d9afa5c6172982ed01","evidence-official-f60303c5eaa60ed9c77a"],"source_locator":"README.md: Model Description, pretrained model list, Application, Installation and licence frontmatter"},{"label":"Known versions","value":"Geneformer-V1-10M, V2-104M, V2-316M and V2-104M_CLcancer; the card states V2-316M is the repository default.","status":"source_checked","source_ids":["evidence-official-d3d9afa5c6172982ed01","evidence-official-f60303c5eaa60ed9c77a"],"source_locator":"README.md: Model Description, pretrained model list, Application, Installation and licence frontmatter"},{"label":"Training data","value":"V1: approximately 30M human single-cell transcriptomes. V2: approximately 104M non-cancer human transcriptomes; the cancer continual-learning variant adds approximately 14M cancer cells.","status":"source_checked","source_ids":["evidence-official-d3d9afa5c6172982ed01","evidence-official-f60303c5eaa60ed9c77a"],"source_locator":"README.md: Model Description, pretrained model list, Application, Installation and licence frontmatter"},{"label":"Training cutoff","value":"Training dates reported: June 2021 for V1 and December 2024 for V2. These are training dates, not independently verified data-collection cutoffs.","status":"source_checked","source_ids":["evidence-official-d3d9afa5c6172982ed01","evidence-official-f60303c5eaa60ed9c77a"],"source_locator":"README.md: Model Description, pretrained model list, Application, Installation and licence frontmatter"},{"label":"Context limits","value":"V1: 2,048 gene tokens; V2: 4,096. Vocabularies also differ.","status":"source_checked","source_ids":["evidence-official-d3d9afa5c6172982ed01","evidence-official-f60303c5eaa60ed9c77a"],"source_locator":"README.md: Model Description, pretrained model list, Application, Installation and licence frontmatter"},{"label":"Weights licence","value":"Apache-2.0 is declared in the official model-card metadata.","status":"source_checked","source_ids":["evidence-official-d3d9afa5c6172982ed01","evidence-official-f60303c5eaa60ed9c77a"],"source_locator":"README.md: Model Description, pretrained model list, Application, Installation and licence frontmatter"},{"label":"Access","value":"Official downloadable model/card and usage examples: https://huggingface.co/ctheodoris/Geneformer","status":"source_checked","source_ids":["evidence-official-d3d9afa5c6172982ed01","evidence-official-f60303c5eaa60ed9c77a"],"source_locator":"README.md: Model Description, pretrained model list, Application, Installation and licence frontmatter"},{"label":"Code licence","value":"The official repository declares Apache-2.0 in its model card. No separate code-licence file appears in the complete inspected revision; this records the repository declaration rather than an independently reviewed licence grant for every bundled dependency.","status":"source_checked","source_ids":["evidence-official-d3d9afa5c6172982ed01","evidence-final-model-geneformer-tree"],"source_locator":"Pinned README license metadata; complete two-page recursive repository inventory at revision 1f7fbae4e469a5f4f1af8c111a529cfe1b3829f5"}],"strengths":[{"text":"Self-supervised pretraining uses unlabeled cells and supports downstream gene/cell classification and in-silico perturbation workflows.","source_ids":["evidence-official-d3d9afa5c6172982ed01","evidence-official-f60303c5eaa60ed9c77a"],"source_locator":"README.md: Model Description, pretrained model list, Application, Installation and licence frontmatter"}],"limitations":[{"text":"V1, V2 and cancer-tuned V2 use different corpora and vocabularies. The authors recommend task-specific hyperparameter tuning; there is no universally suitable fine-tuning configuration.","source_ids":["evidence-official-d3d9afa5c6172982ed01","evidence-official-f60303c5eaa60ed9c77a"],"source_locator":"README.md: Model Description, pretrained model list, Application, Installation and licence frontmatter"}],"diagram":{"title":"Geneformer workflow","steps":["Expression profile","Corpus-normalized gene ranks","Masked-gene transformer","Cell and gene representations"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-d3d9afa5c6172982ed01","evidence-official-f60303c5eaa60ed9c77a"],"source_locator":"README.md: Model Description, pretrained model list, Application, Installation and licence frontmatter"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied. Follow-up retrieved the complete original RNA-FM PDF and inspected the full pinned Geneformer repository inventory; unavailable labels were updated only where new evidence resolved the earlier retrieval gap."}}}} {"id":"catalog-model-kraken2","kind":"model","name":"Kraken2","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["microbes-communities"],"method_types":["baseline"]},"source_ids":["catalog-source-kraken2"],"links":[],"attributes":{"entity_level":"family","version":"current database pinned at run time","reported_name":"Kraken2","access":"Public classifier; database build/version must be pinned separately.","method_type":"baseline","historical_missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"Kraken 2 assigns taxonomic labels to sequence reads by consulting a reference-derived minimizer database.","summary_source_ids":["evidence-official-40f46dc4b9fcc306773b","evidence-official-33fe0374c8ebbc2c7294","evidence-official-9ba2b6dc494c7b6df951","evidence-official-1f756fefa823dd402b0f"],"summary_source_locator":"docs/MANUAL.markdown: Introduction, Standard Kraken 2 Database, Classification and Confidence scoring","sections":[{"title":"How it works","body":"Kraken 2 breaks query sequences into k-mers and looks up selected minimizers in a compact hash table. Each stored minimizer is associated with a lowest-common-ancestor taxonomic label. The classifier combines that evidence to assign a taxon; the reference database and confidence settings are therefore part of the evaluated procedure.","source_ids":["evidence-official-40f46dc4b9fcc306773b","evidence-official-33fe0374c8ebbc2c7294","evidence-official-9ba2b6dc494c7b6df951","evidence-official-1f756fefa823dd402b0f"],"source_locator":"docs/MANUAL.markdown: Introduction, Standard Kraken 2 Database, Classification and Confidence scoring"},{"title":"Versions and reproducibility","body":"Kraken 2 is a rewrite of Kraken 1 and is not backwards compatible. Read/contig input, not a learned fixed token window.","source_ids":["evidence-official-40f46dc4b9fcc306773b","evidence-official-33fe0374c8ebbc2c7294","evidence-official-9ba2b6dc494c7b6df951","evidence-official-1f756fefa823dd402b0f"],"source_locator":"docs/MANUAL.markdown: Introduction, Standard Kraken 2 Database, Classification and Confidence scoring"}],"facts":[{"label":"Model type","value":"Minimizer-based taxonomic classifier","status":"source_checked","source_ids":["evidence-official-40f46dc4b9fcc306773b","evidence-official-33fe0374c8ebbc2c7294","evidence-official-9ba2b6dc494c7b6df951","evidence-official-1f756fefa823dd402b0f"],"source_locator":"docs/MANUAL.markdown: Introduction, Standard Kraken 2 Database, Classification and Confidence scoring"},{"label":"Architecture","value":"Minimizer-based sequence classification using a compact hash table and lowest-common-ancestor taxonomy assignments.","status":"source_checked","source_ids":["evidence-official-40f46dc4b9fcc306773b","evidence-official-33fe0374c8ebbc2c7294","evidence-official-9ba2b6dc494c7b6df951","evidence-official-1f756fefa823dd402b0f"],"source_locator":"docs/MANUAL.markdown: Introduction, Standard Kraken 2 Database, Classification and Confidence scoring"},{"label":"Inputs","value":"DNA reads or, in translated-search mode, sequences searched against an amino-acid database.","status":"source_checked","source_ids":["evidence-official-40f46dc4b9fcc306773b","evidence-official-33fe0374c8ebbc2c7294","evidence-official-9ba2b6dc494c7b6df951","evidence-official-1f756fefa823dd402b0f"],"source_locator":"docs/MANUAL.markdown: Introduction, Standard Kraken 2 Database, Classification and Confidence scoring"},{"label":"Outputs","value":"Per-read taxonomic assignments and aggregate classification reports.","status":"source_checked","source_ids":["evidence-official-40f46dc4b9fcc306773b","evidence-official-33fe0374c8ebbc2c7294","evidence-official-9ba2b6dc494c7b6df951","evidence-official-1f756fefa823dd402b0f"],"source_locator":"docs/MANUAL.markdown: Introduction, Standard Kraken 2 Database, Classification and Confidence scoring"},{"label":"Parameters","value":"Inapplicable as a neural parameter total; k-mer/minimizer length, confidence and database choices are algorithm settings.","status":"inapplicable","source_ids":["evidence-official-40f46dc4b9fcc306773b","evidence-official-33fe0374c8ebbc2c7294","evidence-official-9ba2b6dc494c7b6df951","evidence-official-1f756fefa823dd402b0f"],"source_locator":"docs/MANUAL.markdown: Introduction, Standard Kraken 2 Database, Classification and Confidence scoring"},{"label":"Known versions","value":"Kraken 2 is a rewrite of Kraken 1 and is not backwards compatible.","status":"source_checked","source_ids":["evidence-official-40f46dc4b9fcc306773b","evidence-official-33fe0374c8ebbc2c7294","evidence-official-9ba2b6dc494c7b6df951","evidence-official-1f756fefa823dd402b0f"],"source_locator":"docs/MANUAL.markdown: Introduction, Standard Kraken 2 Database, Classification and Confidence scoring"},{"label":"Training data","value":"Not neural pretraining: build a database from selected reference sequences and taxonomy.","status":"source_checked","source_ids":["evidence-official-40f46dc4b9fcc306773b","evidence-official-33fe0374c8ebbc2c7294","evidence-official-9ba2b6dc494c7b6df951","evidence-official-1f756fefa823dd402b0f"],"source_locator":"docs/MANUAL.markdown: Introduction, Standard Kraken 2 Database, Classification and Confidence scoring"},{"label":"Training cutoff","value":"Inapplicable to neural pretraining; reference-database and input-data dates must be recorded for each run.","status":"inapplicable","source_ids":["evidence-official-40f46dc4b9fcc306773b","evidence-official-33fe0374c8ebbc2c7294","evidence-official-9ba2b6dc494c7b6df951","evidence-official-1f756fefa823dd402b0f"],"source_locator":"docs/MANUAL.markdown: Introduction, Standard Kraken 2 Database, Classification and Confidence scoring"},{"label":"Context limits","value":"Read/contig input, not a learned fixed token window.","status":"source_checked","source_ids":["evidence-official-40f46dc4b9fcc306773b","evidence-official-33fe0374c8ebbc2c7294","evidence-official-9ba2b6dc494c7b6df951","evidence-official-1f756fefa823dd402b0f"],"source_locator":"docs/MANUAL.markdown: Introduction, Standard Kraken 2 Database, Classification and Confidence scoring"},{"label":"Weights licence","value":"Inapplicable to this classifier: database contents and their licences replace neural weights.","status":"inapplicable","source_ids":["evidence-official-40f46dc4b9fcc306773b","evidence-official-33fe0374c8ebbc2c7294","evidence-official-9ba2b6dc494c7b6df951","evidence-official-1f756fefa823dd402b0f"],"source_locator":"docs/MANUAL.markdown: Introduction, Standard Kraken 2 Database, Classification and Confidence scoring"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/DerrickWood/kraken2","status":"source_checked","source_ids":["evidence-official-40f46dc4b9fcc306773b","evidence-official-33fe0374c8ebbc2c7294","evidence-official-9ba2b6dc494c7b6df951","evidence-official-1f756fefa823dd402b0f"],"source_locator":"docs/MANUAL.markdown: Introduction, Standard Kraken 2 Database, Classification and Confidence scoring"},{"label":"Code licence","value":"MIT","status":"source_checked","source_ids":["evidence-official-bf1a5e03cd84873f4b04"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"A reference-based procedural comparator with explicit database construction and confidence settings.","source_ids":["evidence-official-40f46dc4b9fcc306773b","evidence-official-33fe0374c8ebbc2c7294","evidence-official-9ba2b6dc494c7b6df951","evidence-official-1f756fefa823dd402b0f"],"source_locator":"docs/MANUAL.markdown: Introduction, Standard Kraken 2 Database, Classification and Confidence scoring"}],"limitations":[{"text":"Classification depends on the reference database, taxonomy version and minimizer configuration. Compact hashing can introduce false matches; software version alone does not identify a reproducible classifier.","source_ids":["evidence-official-40f46dc4b9fcc306773b","evidence-official-33fe0374c8ebbc2c7294","evidence-official-9ba2b6dc494c7b6df951","evidence-official-1f756fefa823dd402b0f"],"source_locator":"docs/MANUAL.markdown: Introduction, Standard Kraken 2 Database, Classification and Confidence scoring"}],"diagram":{"title":"Kraken2 workflow","steps":["Reference genomes and taxonomy","Minimizer database","Read minimizer lookup","Taxonomic assignment"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-40f46dc4b9fcc306773b","evidence-official-33fe0374c8ebbc2c7294","evidence-official-9ba2b6dc494c7b6df951","evidence-official-1f756fefa823dd402b0f"],"source_locator":"docs/MANUAL.markdown: Introduction, Standard Kraken 2 Database, Classification and Confidence scoring"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}}} {"id":"catalog-model-metagene-1","kind":"model","name":"METAGENE-1","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["microbes-communities"],"method_types":["foundation model"]},"source_ids":["catalog-source-metagene-1"],"links":[],"attributes":{"entity_level":"family","version":"6B","reported_name":"METAGENE-1","access":"Public Apache 2.0 checkpoint; 512-token context and large local memory requirement.","method_type":"foundation model","historical_missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"METAGENE-1 is an autoregressive DNA/RNA sequence model trained on wastewater metagenomic data.","summary_source_ids":["evidence-official-8e894fc6d180746a1808","evidence-official-f71e6c488025a4d82997","evidence-official-a6b46962aae5cfc348c6","evidence-official-b554679ea1503ce3d9f6"],"summary_source_locator":"README.md: Model Overview and Usage; config.json; config.json: model_type, hidden_size, num_hidden_layers, num_attention_heads, vocab_size and sequence-length fields; official pretraining train/config_hub/pretrain/genomicsllama.yml: train.max_seq_length","sections":[{"title":"How it works","body":"METAGENE-1 is an autoregressive DNA/RNA sequence model trained on wastewater metagenomic data. Llama-style autoregressive transformer with 32 layers, width 4,096, 32 attention heads and a 1,024-token vocabulary in the inspected configuration. The documented inputs are DNA or RNA nucleotide sequences. The output consists of sequence generation and representations for downstream metagenomic analyses.","source_ids":["evidence-official-8e894fc6d180746a1808","evidence-official-f71e6c488025a4d82997","evidence-official-a6b46962aae5cfc348c6","evidence-official-b554679ea1503ce3d9f6"],"source_locator":"README.md: Model Overview and Usage; config.json; config.json: model_type, hidden_size, num_hidden_layers, num_attention_heads, vocab_size and sequence-length fields; official pretraining train/config_hub/pretrain/genomicsllama.yml: train.max_seq_length"},{"title":"Versions and reproducibility","body":"METAGENE-1; exact model-card and configuration revision pinned in sources. Configuration fields differ: max_position_embeddings=512 and max_sequence_length=2,048. The effective supported window remains unresolved; neither value is silently promoted to a validated inference limit.","source_ids":["evidence-official-8e894fc6d180746a1808","evidence-official-f71e6c488025a4d82997","evidence-official-a6b46962aae5cfc348c6","evidence-official-b554679ea1503ce3d9f6"],"source_locator":"README.md: Model Overview and Usage; config.json; config.json: model_type, hidden_size, num_hidden_layers, num_attention_heads, vocab_size and sequence-length fields; official pretraining train/config_hub/pretrain/genomicsllama.yml: train.max_seq_length"}],"facts":[{"label":"Model type","value":"Autoregressive metagenomic transformer","status":"source_checked","source_ids":["evidence-official-8e894fc6d180746a1808","evidence-official-f71e6c488025a4d82997","evidence-official-a6b46962aae5cfc348c6","evidence-official-b554679ea1503ce3d9f6"],"source_locator":"README.md: Model Overview and Usage; config.json; config.json: model_type, hidden_size, num_hidden_layers, num_attention_heads, vocab_size and sequence-length fields; official pretraining train/config_hub/pretrain/genomicsllama.yml: train.max_seq_length"},{"label":"Architecture","value":"Llama-style autoregressive transformer with 32 layers, width 4,096, 32 attention heads and a 1,024-token vocabulary in the inspected configuration.","status":"source_checked","source_ids":["evidence-official-8e894fc6d180746a1808","evidence-official-f71e6c488025a4d82997","evidence-official-a6b46962aae5cfc348c6","evidence-official-b554679ea1503ce3d9f6"],"source_locator":"README.md: Model Overview and Usage; config.json; config.json: model_type, hidden_size, num_hidden_layers, num_attention_heads, vocab_size and sequence-length fields; official pretraining train/config_hub/pretrain/genomicsllama.yml: train.max_seq_length"},{"label":"Inputs","value":"DNA or RNA nucleotide sequences.","status":"source_checked","source_ids":["evidence-official-8e894fc6d180746a1808","evidence-official-f71e6c488025a4d82997","evidence-official-a6b46962aae5cfc348c6","evidence-official-b554679ea1503ce3d9f6"],"source_locator":"README.md: Model Overview and Usage; config.json; config.json: model_type, hidden_size, num_hidden_layers, num_attention_heads, vocab_size and sequence-length fields; official pretraining train/config_hub/pretrain/genomicsllama.yml: train.max_seq_length"},{"label":"Outputs","value":"Sequence generation and representations for downstream metagenomic analyses.","status":"source_checked","source_ids":["evidence-official-8e894fc6d180746a1808","evidence-official-f71e6c488025a4d82997","evidence-official-a6b46962aae5cfc348c6","evidence-official-b554679ea1503ce3d9f6"],"source_locator":"README.md: Model Overview and Usage; config.json; config.json: model_type, hidden_size, num_hidden_layers, num_attention_heads, vocab_size and sequence-length fields; official pretraining train/config_hub/pretrain/genomicsllama.yml: train.max_seq_length"},{"label":"Parameters","value":"7 billion.","status":"source_checked","source_ids":["evidence-official-8e894fc6d180746a1808","evidence-official-f71e6c488025a4d82997","evidence-official-a6b46962aae5cfc348c6","evidence-official-b554679ea1503ce3d9f6"],"source_locator":"README.md: Model Overview and Usage; config.json; config.json: model_type, hidden_size, num_hidden_layers, num_attention_heads, vocab_size and sequence-length fields; official pretraining train/config_hub/pretrain/genomicsllama.yml: train.max_seq_length"},{"label":"Known versions","value":"METAGENE-1; exact model-card and configuration revision pinned in sources.","status":"source_checked","source_ids":["evidence-official-8e894fc6d180746a1808","evidence-official-f71e6c488025a4d82997","evidence-official-a6b46962aae5cfc348c6","evidence-official-b554679ea1503ce3d9f6"],"source_locator":"README.md: Model Overview and Usage; config.json; config.json: model_type, hidden_size, num_hidden_layers, num_attention_heads, vocab_size and sequence-length fields; official pretraining train/config_hub/pretrain/genomicsllama.yml: train.max_seq_length"},{"label":"Training data","value":"More than 1.5 trillion base pairs sequenced from human wastewater samples, according to the model card.","status":"source_checked","source_ids":["evidence-official-8e894fc6d180746a1808","evidence-official-f71e6c488025a4d82997","evidence-official-a6b46962aae5cfc348c6","evidence-official-b554679ea1503ce3d9f6"],"source_locator":"README.md: Model Overview and Usage; config.json; config.json: model_type, hidden_size, num_hidden_layers, num_attention_heads, vocab_size and sequence-length fields; official pretraining train/config_hub/pretrain/genomicsllama.yml: train.max_seq_length"},{"label":"Training cutoff","value":"The official pretraining README states that the wastewater corpus is not yet publicly released; a latest sample-collection date is not provided there.","status":"unreported","source_ids":["evidence-official-8e894fc6d180746a1808","evidence-official-f71e6c488025a4d82997","evidence-official-a6b46962aae5cfc348c6","evidence-official-b554679ea1503ce3d9f6"],"source_locator":"README.md: Model Overview and Usage; config.json; config.json: model_type, hidden_size, num_hidden_layers, num_attention_heads, vocab_size and sequence-length fields; official pretraining train/config_hub/pretrain/genomicsllama.yml: train.max_seq_length"},{"label":"Context limits","value":"The released pretraining YAML sets max_seq_length=512. The model config lists max_position_embeddings=512 and max_sequence_length=2,048; these differing configuration fields do not establish a single validated inference maximum.","status":"source_checked","source_ids":["evidence-official-8e894fc6d180746a1808","evidence-official-f71e6c488025a4d82997","evidence-official-a6b46962aae5cfc348c6","evidence-official-b554679ea1503ce3d9f6"],"source_locator":"README.md: Model Overview and Usage; config.json; config.json: model_type, hidden_size, num_hidden_layers, num_attention_heads, vocab_size and sequence-length fields; official pretraining train/config_hub/pretrain/genomicsllama.yml: train.max_seq_length"},{"label":"Weights licence","value":"Apache-2.0 declared in the model card.","status":"source_checked","source_ids":["evidence-official-8e894fc6d180746a1808","evidence-official-f71e6c488025a4d82997","evidence-official-a6b46962aae5cfc348c6","evidence-official-b554679ea1503ce3d9f6"],"source_locator":"README.md: Model Overview and Usage; config.json; config.json: model_type, hidden_size, num_hidden_layers, num_attention_heads, vocab_size and sequence-length fields; official pretraining train/config_hub/pretrain/genomicsllama.yml: train.max_seq_length"},{"label":"Access","value":"Official downloadable model/card and usage examples: https://huggingface.co/metagene-ai/METAGENE-1","status":"source_checked","source_ids":["evidence-official-8e894fc6d180746a1808","evidence-official-f71e6c488025a4d82997","evidence-official-a6b46962aae5cfc348c6","evidence-official-b554679ea1503ce3d9f6"],"source_locator":"README.md: Model Overview and Usage; config.json; config.json: model_type, hidden_size, num_hidden_layers, num_attention_heads, vocab_size and sequence-length fields; official pretraining train/config_hub/pretrain/genomicsllama.yml: train.max_seq_length"},{"label":"Code licence","value":"Apache-2.0","status":"source_checked","source_ids":["evidence-official-c2dd029568d234cb0d16"],"source_locator":"train/LICENSE: licence text"}],"strengths":[{"text":"Its pretraining corpus includes diverse mixed-community DNA and RNA rather than a single reference genome.","source_ids":["evidence-official-8e894fc6d180746a1808","evidence-official-f71e6c488025a4d82997","evidence-official-a6b46962aae5cfc348c6","evidence-official-b554679ea1503ce3d9f6"],"source_locator":"README.md: Model Overview and Usage; config.json; config.json: model_type, hidden_size, num_hidden_layers, num_attention_heads, vocab_size and sequence-length fields; official pretraining train/config_hub/pretrain/genomicsllama.yml: train.max_seq_length"}],"limitations":[{"text":"The stated biosurveillance and pathogen-detection applications require their own downstream evaluation; the pretraining objective is sequence modeling.","source_ids":["evidence-official-8e894fc6d180746a1808","evidence-official-f71e6c488025a4d82997","evidence-official-a6b46962aae5cfc348c6","evidence-official-b554679ea1503ce3d9f6"],"source_locator":"README.md: Model Overview and Usage; config.json; config.json: model_type, hidden_size, num_hidden_layers, num_attention_heads, vocab_size and sequence-length fields; official pretraining train/config_hub/pretrain/genomicsllama.yml: train.max_seq_length"}],"diagram":{"title":"METAGENE-1 workflow","steps":["Nucleotide sequence","BPE tokens","Autoregressive transformer","Sequence or representations"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-8e894fc6d180746a1808","evidence-official-f71e6c488025a4d82997","evidence-official-a6b46962aae5cfc348c6","evidence-official-b554679ea1503ce3d9f6"],"source_locator":"README.md: Model Overview and Usage; config.json; config.json: model_type, hidden_size, num_hidden_layers, num_attention_heads, vocab_size and sequence-length fields; official pretraining train/config_hub/pretrain/genomicsllama.yml: train.max_seq_length"},"coverage":"limited","gaps":["Training cutoff: The official pretraining README states that the wastewater corpus is not yet publicly released; a latest sample-collection date is not provided there."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}}} {"id":"catalog-model-metaphlan","kind":"model","name":"MetaPhlAn","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["microbes-communities"],"method_types":["specialist"]},"source_ids":["catalog-source-metaphlan"],"links":[],"attributes":{"entity_level":"family","version":"current marker database pinned at run time","reported_name":"MetaPhlAn","access":"Public profiler; marker database version must be pinned separately.","method_type":"specialist","historical_missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"MetaPhlAn profiles microbial community composition from shotgun metagenomic reads using clade-specific marker genes.","summary_source_ids":["evidence-official-72fdaed158e4ff851fbe","evidence-official-37f825d1226e59cbbfbf"],"summary_source_locator":"README.md: description; MetaPhlAn 4 paper: Building the expanded SGB catalog and Discussion","sections":[{"title":"How it works","body":"MetaPhlAn profiles microbial community composition from shotgun metagenomic reads using clade-specific marker genes. Reference marker-gene profiling; MetaPhlAn 4 organizes reference and metagenome-assembled genomes into species-level genome bins. The documented inputs are shotgun metagenomic reads and a selected MetaPhlAn marker database. The output consists of taxonomic relative-abundance profiles; StrainPhlAn is a separate strain-level analysis.","source_ids":["evidence-official-72fdaed158e4ff851fbe","evidence-official-37f825d1226e59cbbfbf"],"source_locator":"README.md: description; MetaPhlAn 4 paper: Building the expanded SGB catalog and Discussion"},{"title":"Versions and reproducibility","body":"MetaPhlAn 4 paper and 4.2-linked current documentation; database version is a separate reproducibility requirement. Shotgun reads; no fixed neural token context.","source_ids":["evidence-official-72fdaed158e4ff851fbe","evidence-official-37f825d1226e59cbbfbf"],"source_locator":"README.md: description; MetaPhlAn 4 paper: Building the expanded SGB catalog and Discussion"}],"facts":[{"label":"Model type","value":"Marker-based taxonomic profiling","status":"source_checked","source_ids":["evidence-official-72fdaed158e4ff851fbe","evidence-official-37f825d1226e59cbbfbf"],"source_locator":"README.md: description; MetaPhlAn 4 paper: Building the expanded SGB catalog and Discussion"},{"label":"Architecture","value":"Reference marker-gene profiling; MetaPhlAn 4 organizes reference and metagenome-assembled genomes into species-level genome bins.","status":"source_checked","source_ids":["evidence-official-72fdaed158e4ff851fbe","evidence-official-37f825d1226e59cbbfbf"],"source_locator":"README.md: description; MetaPhlAn 4 paper: Building the expanded SGB catalog and Discussion"},{"label":"Inputs","value":"Shotgun metagenomic reads and a selected MetaPhlAn marker database.","status":"source_checked","source_ids":["evidence-official-72fdaed158e4ff851fbe","evidence-official-37f825d1226e59cbbfbf"],"source_locator":"README.md: description; MetaPhlAn 4 paper: Building the expanded SGB catalog and Discussion"},{"label":"Outputs","value":"Taxonomic relative-abundance profiles; StrainPhlAn is a separate strain-level analysis.","status":"source_checked","source_ids":["evidence-official-72fdaed158e4ff851fbe","evidence-official-37f825d1226e59cbbfbf"],"source_locator":"README.md: description; MetaPhlAn 4 paper: Building the expanded SGB catalog and Discussion"},{"label":"Parameters","value":"Inapplicable as a neural parameter count.","status":"inapplicable","source_ids":["evidence-official-72fdaed158e4ff851fbe","evidence-official-37f825d1226e59cbbfbf"],"source_locator":"README.md: description; MetaPhlAn 4 paper: Building the expanded SGB catalog and Discussion"},{"label":"Known versions","value":"MetaPhlAn 4 paper and 4.2-linked current documentation; database version is a separate reproducibility requirement.","status":"source_checked","source_ids":["evidence-official-72fdaed158e4ff851fbe","evidence-official-37f825d1226e59cbbfbf"],"source_locator":"README.md: description; MetaPhlAn 4 paper: Building the expanded SGB catalog and Discussion"},{"label":"Training data","value":"Reference-derived marker database rather than neural pretraining; record the exact database release.","status":"source_checked","source_ids":["evidence-official-72fdaed158e4ff851fbe","evidence-official-37f825d1226e59cbbfbf"],"source_locator":"README.md: description; MetaPhlAn 4 paper: Building the expanded SGB catalog and Discussion"},{"label":"Training cutoff","value":"Inapplicable to neural pretraining; reference-database and input-data dates must be recorded for each run.","status":"inapplicable","source_ids":["evidence-official-72fdaed158e4ff851fbe","evidence-official-37f825d1226e59cbbfbf"],"source_locator":"README.md: description; MetaPhlAn 4 paper: Building the expanded SGB catalog and Discussion"},{"label":"Context limits","value":"Shotgun reads; no fixed neural token context.","status":"source_checked","source_ids":["evidence-official-72fdaed158e4ff851fbe","evidence-official-37f825d1226e59cbbfbf"],"source_locator":"README.md: description; MetaPhlAn 4 paper: Building the expanded SGB catalog and Discussion"},{"label":"Weights licence","value":"Inapplicable to this procedural method; marker databases have their own provenance and terms.","status":"inapplicable","source_ids":["evidence-official-72fdaed158e4ff851fbe","evidence-official-37f825d1226e59cbbfbf"],"source_locator":"README.md: description; MetaPhlAn 4 paper: Building the expanded SGB catalog and Discussion"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/biobakery/MetaPhlAn","status":"source_checked","source_ids":["evidence-official-72fdaed158e4ff851fbe","evidence-official-37f825d1226e59cbbfbf"],"source_locator":"README.md: description; MetaPhlAn 4 paper: Building the expanded SGB catalog and Discussion"},{"label":"Code licence","value":"MIT","status":"source_checked","source_ids":["evidence-official-1a0775cac85be75449ef"],"source_locator":"license.txt: licence text"}],"strengths":[{"text":"Marker-based profiling can incorporate characterized and previously uncharacterized species groups.","source_ids":["evidence-official-72fdaed158e4ff851fbe","evidence-official-37f825d1226e59cbbfbf"],"source_locator":"README.md: description; MetaPhlAn 4 paper: Building the expanded SGB catalog and Discussion"}],"limitations":[{"text":"Coverage depends on the marker database and habitat. The MetaPhlAn 4 paper identifies remaining gaps for under-studied environmental communities; newer software/databases may have different scope.","source_ids":["evidence-official-72fdaed158e4ff851fbe","evidence-official-37f825d1226e59cbbfbf"],"source_locator":"README.md: description; MetaPhlAn 4 paper: Building the expanded SGB catalog and Discussion"}],"diagram":{"title":"MetaPhlAn workflow","steps":["Metagenomic reads","Marker-gene mapping","Species-group quantification","Relative-abundance profile"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-72fdaed158e4ff851fbe","evidence-official-37f825d1226e59cbbfbf"],"source_locator":"README.md: description; MetaPhlAn 4 paper: Building the expanded SGB catalog and Discussion"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}}} {"id":"catalog-model-mimic","kind":"model","name":"MIMIC","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["rna-transcriptomes","proteins-complexes"],"method_types":["foundation model"]},"source_ids":["catalog-source-mimic"],"links":[],"attributes":{"entity_level":"family","version":"1.0","reported_name":"MIMIC","access":"Public MIT code and 1.25B-parameter weights; large local memory requirement.","method_type":"foundation model","historical_missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"MIMIC represents DNA, RNA, protein and associated molecular measurements in a shared multimodal model.","summary_source_ids":["evidence-official-af8da643e933eae36a68","evidence-official-4565e66fdeb787b99afa","evidence-official-c14c0e6a0fdbe1479658","evidence-official-0ffe1031786a71400ffd"],"summary_source_locator":"MIMIC paper Section 3 and Discussion; official model-card Architecture and License; config.json num_input_tokens and num_target_tokens","sections":[{"title":"How it works","body":"MIMIC represents DNA, RNA, protein and associated molecular measurements in a shared multimodal model. Transformer encoder-decoder: 20 encoder layers and 12 decoder layers, width 1,536, rotary positional embeddings, mixed attention and five register tokens. The documented inputs are co-observed molecular modalities and optional text context, grouped into nucleic, protein and text tracks. The output consists of embeddings or generated modalities, such as residue solvent accessibility or splice-site classes.","source_ids":["evidence-official-af8da643e933eae36a68","evidence-official-4565e66fdeb787b99afa","evidence-official-c14c0e6a0fdbe1479658","evidence-official-0ffe1031786a71400ffd"],"source_locator":"MIMIC paper Section 3 and Discussion; official model-card Architecture and License; config.json num_input_tokens and num_target_tokens"},{"title":"Versions and reproducibility","body":"MIMIC 1.0; load_pretrained(version=\"1.0\") example. Training curriculum increases encoder context from 1k to 10k tokens; the released configuration specifies 10,000 input tokens and 1,000 target tokens.","source_ids":["evidence-official-af8da643e933eae36a68","evidence-official-4565e66fdeb787b99afa","evidence-official-c14c0e6a0fdbe1479658","evidence-official-0ffe1031786a71400ffd"],"source_locator":"MIMIC paper Section 3 and Discussion; official model-card Architecture and License; config.json num_input_tokens and num_target_tokens"}],"facts":[{"label":"Model type","value":"Multimodal biomolecular transformer encoder-decoder","status":"source_checked","source_ids":["evidence-official-af8da643e933eae36a68","evidence-official-4565e66fdeb787b99afa","evidence-official-c14c0e6a0fdbe1479658","evidence-official-0ffe1031786a71400ffd"],"source_locator":"MIMIC paper Section 3 and Discussion; official model-card Architecture and License; config.json num_input_tokens and num_target_tokens"},{"label":"Architecture","value":"Transformer encoder-decoder: 20 encoder layers and 12 decoder layers, width 1,536, rotary positional embeddings, mixed attention and five register tokens.","status":"source_checked","source_ids":["evidence-official-af8da643e933eae36a68","evidence-official-4565e66fdeb787b99afa","evidence-official-c14c0e6a0fdbe1479658","evidence-official-0ffe1031786a71400ffd"],"source_locator":"MIMIC paper Section 3 and Discussion; official model-card Architecture and License; config.json num_input_tokens and num_target_tokens"},{"label":"Inputs","value":"Co-observed molecular modalities and optional text context, grouped into nucleic, protein and text tracks.","status":"source_checked","source_ids":["evidence-official-af8da643e933eae36a68","evidence-official-4565e66fdeb787b99afa","evidence-official-c14c0e6a0fdbe1479658","evidence-official-0ffe1031786a71400ffd"],"source_locator":"MIMIC paper Section 3 and Discussion; official model-card Architecture and License; config.json num_input_tokens and num_target_tokens"},{"label":"Outputs","value":"Embeddings or generated modalities, such as residue solvent accessibility or splice-site classes.","status":"source_checked","source_ids":["evidence-official-af8da643e933eae36a68","evidence-official-4565e66fdeb787b99afa","evidence-official-c14c0e6a0fdbe1479658","evidence-official-0ffe1031786a71400ffd"],"source_locator":"MIMIC paper Section 3 and Discussion; official model-card Architecture and License; config.json num_input_tokens and num_target_tokens"},{"label":"Parameters","value":"Approximately 1.25 billion.","status":"source_checked","source_ids":["evidence-official-af8da643e933eae36a68","evidence-official-4565e66fdeb787b99afa","evidence-official-c14c0e6a0fdbe1479658","evidence-official-0ffe1031786a71400ffd"],"source_locator":"MIMIC paper Section 3 and Discussion; official model-card Architecture and License; config.json num_input_tokens and num_target_tokens"},{"label":"Known versions","value":"MIMIC 1.0; load_pretrained(version=\"1.0\") example.","status":"source_checked","source_ids":["evidence-official-af8da643e933eae36a68","evidence-official-4565e66fdeb787b99afa","evidence-official-c14c0e6a0fdbe1479658","evidence-official-0ffe1031786a71400ffd"],"source_locator":"MIMIC paper Section 3 and Discussion; official model-card Architecture and License; config.json num_input_tokens and num_target_tokens"},{"label":"Training data","value":"LORE aligns approximately 15.5M proteins and 13M RNA transcripts from over 6,000 organisms with associated molecular measurements and over 4B text tokens. Missing modalities are retained as partially observed examples.","status":"source_checked","source_ids":["evidence-official-af8da643e933eae36a68","evidence-official-4565e66fdeb787b99afa","evidence-official-c14c0e6a0fdbe1479658","evidence-official-0ffe1031786a71400ffd"],"source_locator":"MIMIC paper Section 3 and Discussion; official model-card Architecture and License; config.json num_input_tokens and num_target_tokens"},{"label":"Training cutoff","value":"LORE protein/transcript linking uses UniProt release 2024_04. This component version does not establish a common latest-data date for all molecular and text tracks.","status":"source_checked","source_ids":["evidence-official-af8da643e933eae36a68","evidence-official-4565e66fdeb787b99afa","evidence-official-c14c0e6a0fdbe1479658","evidence-official-0ffe1031786a71400ffd"],"source_locator":"MIMIC paper Section 3 and Discussion; official model-card Architecture and License; config.json num_input_tokens and num_target_tokens"},{"label":"Context limits","value":"Training curriculum increases encoder context from 1k to 10k tokens; the released configuration specifies 10,000 input tokens and 1,000 target tokens.","status":"source_checked","source_ids":["evidence-official-af8da643e933eae36a68","evidence-official-4565e66fdeb787b99afa","evidence-official-c14c0e6a0fdbe1479658","evidence-official-0ffe1031786a71400ffd"],"source_locator":"MIMIC paper Section 3 and Discussion; official model-card Architecture and License; config.json num_input_tokens and num_target_tokens"},{"label":"Weights licence","value":"MIT; model card explicitly covers both model and source code.","status":"source_checked","source_ids":["evidence-official-af8da643e933eae36a68","evidence-official-4565e66fdeb787b99afa","evidence-official-c14c0e6a0fdbe1479658","evidence-official-0ffe1031786a71400ffd"],"source_locator":"MIMIC paper Section 3 and Discussion; official model-card Architecture and License; config.json num_input_tokens and num_target_tokens"},{"label":"Access","value":"Official downloadable model/card and usage examples: https://huggingface.co/polymathic-ai/MIMIC","status":"source_checked","source_ids":["evidence-official-af8da643e933eae36a68","evidence-official-4565e66fdeb787b99afa","evidence-official-c14c0e6a0fdbe1479658","evidence-official-0ffe1031786a71400ffd"],"source_locator":"MIMIC paper Section 3 and Discussion; official model-card Architecture and License; config.json num_input_tokens and num_target_tokens"},{"label":"Code licence","value":"MIT","status":"source_checked","source_ids":["evidence-official-34faacf1ad53d6aa6bef"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"The interface accepts partially observed modality sets and supports both embedding and conditional generation.","source_ids":["evidence-official-af8da643e933eae36a68","evidence-official-4565e66fdeb787b99afa","evidence-official-c14c0e6a0fdbe1479658","evidence-official-0ffe1031786a71400ffd"],"source_locator":"MIMIC paper Section 3 and Discussion; official model-card Architecture and License; config.json num_input_tokens and num_target_tokens"}],"limitations":[{"text":"The paper identifies uneven modality coverage, missing biological measurements and limited encoder/decoder windows. Predictions for a missing modality remain model inferences rather than experimental observations.","source_ids":["evidence-official-af8da643e933eae36a68","evidence-official-4565e66fdeb787b99afa","evidence-official-c14c0e6a0fdbe1479658","evidence-official-0ffe1031786a71400ffd"],"source_locator":"MIMIC paper Section 3 and Discussion; official model-card Architecture and License; config.json num_input_tokens and num_target_tokens"}],"diagram":{"title":"MIMIC workflow","steps":["Observed modalities","Shared encoder","Track-aware decoder","Embeddings or generated tracks"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-af8da643e933eae36a68","evidence-official-4565e66fdeb787b99afa","evidence-official-c14c0e6a0fdbe1479658","evidence-official-0ffe1031786a71400ffd"],"source_locator":"MIMIC paper Section 3 and Discussion; official model-card Architecture and License; config.json num_input_tokens and num_target_tokens"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}}} {"id":"catalog-model-mrna-fm","kind":"model","name":"mRNA-FM","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["rna-transcriptomes"],"method_types":["foundation model"]},"source_ids":["catalog-source-mrna-fm"],"links":[{"relation":"variant_of","target_id":"discovery-model-rna-fm"}],"attributes":{"entity_level":"family","version":"codon-tokenised","reported_name":"mRNA-FM","access":"Public checkpoint trained on coding sequences (CDS); input must be codon aligned. UTR-only sequences are outside its training modality.","method_type":"foundation model","historical_missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"mRNA-FM encodes coding RNA with codon-level tokens to produce representations for downstream analysis.","summary_source_ids":["evidence-official-fd8e332abdf04a75195b","evidence-official-e17e3864c464d981afc7"],"summary_source_locator":"README.md: Foundation Models table and mRNA-FM quick-start example","sections":[{"title":"How it works","body":"mRNA-FM encodes coding RNA with codon-level tokens to produce representations for downstream analysis. 12-layer transformer encoder with hidden width 1,280 and codon-level input tokens. The documented inputs are coding RNA in the correct reading frame, with sequence length divisible by three. The output consists of contextual token embeddings for a specified downstream RNA task.","source_ids":["evidence-official-fd8e332abdf04a75195b","evidence-official-e17e3864c464d981afc7"],"source_locator":"README.md: Foundation Models table and mRNA-FM quick-start example"},{"title":"Versions and reproducibility","body":"mrna_fm_t12; distinct from base-tokenized rna_fm_t12. The applicable input limits require configuration-specific checking.","source_ids":["evidence-official-fd8e332abdf04a75195b","evidence-official-e17e3864c464d981afc7"],"source_locator":"README.md: Foundation Models table and mRNA-FM quick-start example"}],"facts":[{"label":"Model type","value":"Codon-token RNA transformer encoder","status":"source_checked","source_ids":["evidence-official-fd8e332abdf04a75195b","evidence-official-e17e3864c464d981afc7"],"source_locator":"README.md: Foundation Models table and mRNA-FM quick-start example"},{"label":"Architecture","value":"12-layer transformer encoder with hidden width 1,280 and codon-level input tokens.","status":"source_checked","source_ids":["evidence-official-fd8e332abdf04a75195b","evidence-official-e17e3864c464d981afc7"],"source_locator":"README.md: Foundation Models table and mRNA-FM quick-start example"},{"label":"Inputs","value":"Coding RNA in the correct reading frame, with sequence length divisible by three.","status":"source_checked","source_ids":["evidence-official-fd8e332abdf04a75195b","evidence-official-e17e3864c464d981afc7"],"source_locator":"README.md: Foundation Models table and mRNA-FM quick-start example"},{"label":"Outputs","value":"Contextual token embeddings for a specified downstream RNA task.","status":"source_checked","source_ids":["evidence-official-fd8e332abdf04a75195b","evidence-official-e17e3864c464d981afc7"],"source_locator":"README.md: Foundation Models table and mRNA-FM quick-start example"},{"label":"Parameters","value":"239M, as printed in the official Foundation Models table.","status":"source_checked","source_ids":["evidence-official-fd8e332abdf04a75195b","evidence-official-e17e3864c464d981afc7"],"source_locator":"README.md: Foundation Models table and mRNA-FM quick-start example"},{"label":"Known versions","value":"mrna_fm_t12; distinct from base-tokenized rna_fm_t12.","status":"source_checked","source_ids":["evidence-official-fd8e332abdf04a75195b","evidence-official-e17e3864c464d981afc7"],"source_locator":"README.md: Foundation Models table and mRNA-FM quick-start example"},{"label":"Training data","value":"45M messenger RNA sequences, as printed in the official Foundation Models table.","status":"source_checked","source_ids":["evidence-official-fd8e332abdf04a75195b","evidence-official-e17e3864c464d981afc7"],"source_locator":"README.md: Foundation Models table and mRNA-FM quick-start example"},{"label":"Training cutoff","value":"The official model table reports 45M coding RNAs but does not identify a latest-data date for that separate mRNA-FM corpus.","status":"unreported","source_ids":["evidence-official-fd8e332abdf04a75195b","evidence-official-e17e3864c464d981afc7"],"source_locator":"README.md: Foundation Models table and mRNA-FM quick-start example"},{"label":"Context limits","value":"The inspected mRNA-FM usage example and loader do not specify a validated maximum codon-sequence length. The retrieved original RNA-FM paper describes the nucleotide-based non-coding RNA model; its limit cannot establish the separate mRNA-FM checkpoint limit.","status":"unreported","source_ids":["evidence-official-fd8e332abdf04a75195b","evidence-official-e17e3864c464d981afc7","evidence-final-model-rna-fm-paper"],"source_locator":"Official README: Foundation Models and mRNA-FM quick start; loader; original RNA-FM paper Methods, pp.22–23"},{"label":"Weights licence","value":"Separate checkpoint-distribution terms are not stated in the inspected release documentation and licence material. The source-code licence alone is not recorded as an explicit weight grant.","status":"unreported","source_ids":["evidence-official-fd8e332abdf04a75195b","evidence-official-e17e3864c464d981afc7","evidence-official-3fde3df73e79e455bd86"],"source_locator":"README.md: Foundation Models table and mRNA-FM quick-start example; LICENSE: licence text"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/ml4bio/RNA-FM","status":"source_checked","source_ids":["evidence-official-fd8e332abdf04a75195b","evidence-official-e17e3864c464d981afc7"],"source_locator":"README.md: Foundation Models table and mRNA-FM quick-start example"},{"label":"Code licence","value":"MIT","status":"source_checked","source_ids":["evidence-official-3fde3df73e79e455bd86"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"The repository exposes embedding extraction and examples for downstream RNA analyses.","source_ids":["evidence-official-fd8e332abdf04a75195b","evidence-official-e17e3864c464d981afc7"],"source_locator":"README.md: Foundation Models table and mRNA-FM quick-start example"}],"limitations":[{"text":"Base-level RNA-FM and codon-level mRNA-FM are not interchangeable. The task head and tokenization must be specified in each evaluation.","source_ids":["evidence-official-fd8e332abdf04a75195b","evidence-official-e17e3864c464d981afc7"],"source_locator":"README.md: Foundation Models table and mRNA-FM quick-start example"}],"diagram":{"title":"mRNA-FM workflow","steps":["Coding RNA","Codon tokenizer","12-layer transformer encoder","Contextual codon representations"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-fd8e332abdf04a75195b","evidence-official-e17e3864c464d981afc7"],"source_locator":"README.md: Foundation Models table and mRNA-FM quick-start example"},"coverage":"limited","gaps":["Training cutoff: The official model table reports 45M coding RNAs but does not identify a latest-data date for that separate mRNA-FM corpus.","Context limits: The inspected mRNA-FM usage example and loader do not specify a validated maximum codon-sequence length. The retrieved original RNA-FM paper describes the nucleotide-based non-coding RNA model; its limit cannot establish the separate mRNA-FM checkpoint limit.","Weights licence: Separate checkpoint-distribution terms are not stated in the inspected release documentation and licence material. The source-code licence alone is not recorded as an explicit weight grant."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied. Follow-up retrieved the complete original RNA-FM PDF and inspected the full pinned Geneformer repository inventory; unavailable labels were updated only where new evidence resolved the earlier retrieval gap."}}}} {"id":"catalog-model-nt-v2","kind":"model","name":"Nucleotide Transformer v2","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["dna-genomes"],"method_types":["foundation model"]},"source_ids":["catalog-source-nt-v2"],"links":[],"attributes":{"entity_level":"family","version":"50M multi-species","reported_name":"Nucleotide Transformer v2","access":"Public checkpoint.","method_type":"foundation model","historical_missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"Nucleotide Transformer v2 represents DNA using an encoder pretrained on multiple species.","summary_source_ids":["evidence-official-4abb9affb1a9e5438c91","evidence-official-6b79ddfdd330693bf4fb","evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-aadbeb0f10ec55d34f5f"],"summary_source_locator":"README.md: Model Summary, Training data and licence metadata; config.json; Nucleotide Transformer paper Methods: Architecture and Training (v2); source conflict with 50M card retained","sections":[{"title":"How it works","body":"Nucleotide Transformer v2 represents DNA using an encoder pretrained on multiple species. Encoder-only transformer with 6-mer tokenization, rotary position embeddings and SwiGLU feed-forward layers. The documented inputs are DNA sequences with tokenization determined by 6-mers and individual ambiguous/remainder bases. The output consists of contextual DNA embeddings and masked-token probabilities; downstream tasks need adaptation.","source_ids":["evidence-official-4abb9affb1a9e5438c91","evidence-official-6b79ddfdd330693bf4fb","evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-aadbeb0f10ec55d34f5f"],"source_locator":"README.md: Model Summary, Training data and licence metadata; config.json; Nucleotide Transformer paper Methods: Architecture and Training (v2); source conflict with 50M card retained"},{"title":"Versions and reproducibility","body":"nucleotide-transformer-v2-50m-multi-species is the linked checkpoint; distinguish it from other family scales. Source conflict retained: the model card describes 1,000-token pretraining, while the official NT-v2 documentation describes 2,048-token capacity. Token and base counts must be stated separately.","source_ids":["evidence-official-4abb9affb1a9e5438c91","evidence-official-6b79ddfdd330693bf4fb","evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-aadbeb0f10ec55d34f5f"],"source_locator":"README.md: Model Summary, Training data and licence metadata; config.json; Nucleotide Transformer paper Methods: Architecture and Training (v2); source conflict with 50M card retained"}],"facts":[{"label":"Model type","value":"DNA transformer encoder","status":"source_checked","source_ids":["evidence-official-4abb9affb1a9e5438c91","evidence-official-6b79ddfdd330693bf4fb","evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-aadbeb0f10ec55d34f5f"],"source_locator":"README.md: Model Summary, Training data and licence metadata; config.json; Nucleotide Transformer paper Methods: Architecture and Training (v2); source conflict with 50M card retained"},{"label":"Architecture","value":"Encoder-only transformer with 6-mer tokenization, rotary position embeddings and SwiGLU feed-forward layers.","status":"source_checked","source_ids":["evidence-official-4abb9affb1a9e5438c91","evidence-official-6b79ddfdd330693bf4fb","evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-aadbeb0f10ec55d34f5f"],"source_locator":"README.md: Model Summary, Training data and licence metadata; config.json; Nucleotide Transformer paper Methods: Architecture and Training (v2); source conflict with 50M card retained"},{"label":"Inputs","value":"DNA sequences with tokenization determined by 6-mers and individual ambiguous/remainder bases.","status":"source_checked","source_ids":["evidence-official-4abb9affb1a9e5438c91","evidence-official-6b79ddfdd330693bf4fb","evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-aadbeb0f10ec55d34f5f"],"source_locator":"README.md: Model Summary, Training data and licence metadata; config.json; Nucleotide Transformer paper Methods: Architecture and Training (v2); source conflict with 50M card retained"},{"label":"Outputs","value":"Contextual DNA embeddings and masked-token probabilities; downstream tasks need adaptation.","status":"source_checked","source_ids":["evidence-official-4abb9affb1a9e5438c91","evidence-official-6b79ddfdd330693bf4fb","evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-aadbeb0f10ec55d34f5f"],"source_locator":"README.md: Model Summary, Training data and licence metadata; config.json; Nucleotide Transformer paper Methods: Architecture and Training (v2); source conflict with 50M card retained"},{"label":"Parameters","value":"The linked checkpoint is 50M; v2 family also includes 100M, 250M and 500M.","status":"source_checked","source_ids":["evidence-official-4abb9affb1a9e5438c91","evidence-official-6b79ddfdd330693bf4fb","evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-aadbeb0f10ec55d34f5f"],"source_locator":"README.md: Model Summary, Training data and licence metadata; config.json; Nucleotide Transformer paper Methods: Architecture and Training (v2); source conflict with 50M card retained"},{"label":"Known versions","value":"nucleotide-transformer-v2-50m-multi-species is the linked checkpoint; distinguish it from other family scales.","status":"source_checked","source_ids":["evidence-official-4abb9affb1a9e5438c91","evidence-official-6b79ddfdd330693bf4fb","evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-aadbeb0f10ec55d34f5f"],"source_locator":"README.md: Model Summary, Training data and licence metadata; config.json; Nucleotide Transformer paper Methods: Architecture and Training (v2); source conflict with 50M card retained"},{"label":"Training data","value":"The linked 50M card reports 850 reference genomes, excluding plants and viruses; 174B source nucleotides and 300B training tokens. Source corpus size is distinct from repeated training exposure.","status":"source_checked","source_ids":["evidence-official-4abb9affb1a9e5438c91","evidence-official-6b79ddfdd330693bf4fb","evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-aadbeb0f10ec55d34f5f"],"source_locator":"README.md: Model Summary, Training data and licence metadata; config.json; Nucleotide Transformer paper Methods: Architecture and Training (v2); source conflict with 50M card retained"},{"label":"Training cutoff","value":"The paper specifies human-reference, 1000 Genomes and multispecies training collections by variant. A single latest-deposition date for all sequences is not supplied in the inspected pretraining-data section.","status":"unreported","source_ids":["evidence-official-4abb9affb1a9e5438c91","evidence-official-6b79ddfdd330693bf4fb","evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-aadbeb0f10ec55d34f5f"],"source_locator":"README.md: Model Summary, Training data and licence metadata; config.json; Nucleotide Transformer paper Methods: Architecture and Training (v2); source conflict with 50M card retained"},{"label":"Context limits","value":"The inspected 50M model card describes 1,000-token pretraining, whereas the NT-v2 paper and documentation describe 2,048-token capacity. Preserve this source discrepancy and the selected checkpoint; tokens and bases differ.","status":"source_checked","source_ids":["evidence-official-4abb9affb1a9e5438c91","evidence-official-6b79ddfdd330693bf4fb","evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-aadbeb0f10ec55d34f5f"],"source_locator":"README.md: Model Summary, Training data and licence metadata; config.json; Nucleotide Transformer paper Methods: Architecture and Training (v2); source conflict with 50M card retained"},{"label":"Weights licence","value":"CC-BY-NC-SA-4.0 as declared in the official model card.","status":"source_checked","source_ids":["evidence-official-4abb9affb1a9e5438c91","evidence-official-6b79ddfdd330693bf4fb","evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-aadbeb0f10ec55d34f5f"],"source_locator":"README.md: Model Summary, Training data and licence metadata; config.json; Nucleotide Transformer paper Methods: Architecture and Training (v2); source conflict with 50M card retained"},{"label":"Access","value":"Official downloadable model/card and usage examples: https://huggingface.co/InstaDeepAI/nucleotide-transformer-v2-50m-multi-species","status":"source_checked","source_ids":["evidence-official-4abb9affb1a9e5438c91","evidence-official-6b79ddfdd330693bf4fb","evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-aadbeb0f10ec55d34f5f"],"source_locator":"README.md: Model Summary, Training data and licence metadata; config.json; Nucleotide Transformer paper Methods: Architecture and Training (v2); source conflict with 50M card retained"},{"label":"Code licence","value":"CC-BY-NC-SA-4.0","status":"source_checked","source_ids":["evidence-official-7e4b193e47ba209860a1"],"source_locator":"LICENSE.md: licence text"}],"strengths":[{"text":"The v2 architecture extends the token context relative to v1 while providing several parameter scales.","source_ids":["evidence-official-4abb9affb1a9e5438c91","evidence-official-6b79ddfdd330693bf4fb","evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-aadbeb0f10ec55d34f5f"],"source_locator":"README.md: Model Summary, Training data and licence metadata; config.json; Nucleotide Transformer paper Methods: Architecture and Training (v2); source conflict with 50M card retained"}],"limitations":[{"text":"Token length is not identical to base-pair length: ambiguous bases consume individual tokens. The linked 50M card does not make every v2 checkpoint identical.","source_ids":["evidence-official-4abb9affb1a9e5438c91","evidence-official-6b79ddfdd330693bf4fb","evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-aadbeb0f10ec55d34f5f"],"source_locator":"README.md: Model Summary, Training data and licence metadata; config.json; Nucleotide Transformer paper Methods: Architecture and Training (v2); source conflict with 50M card retained"}],"diagram":{"title":"Nucleotide Transformer v2 workflow","steps":["DNA sequence","6-mer tokenizer","NT-v2 encoder","Embeddings or task adaptation"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-4abb9affb1a9e5438c91","evidence-official-6b79ddfdd330693bf4fb","evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-aadbeb0f10ec55d34f5f"],"source_locator":"README.md: Model Summary, Training data and licence metadata; config.json; Nucleotide Transformer paper Methods: Architecture and Training (v2); source conflict with 50M card retained"},"coverage":"limited","gaps":["Training cutoff: The paper specifies human-reference, 1000 Genomes and multispecies training collections by variant. A single latest-deposition date for all sequences is not supplied in the inspected pretraining-data section."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}}} {"id":"catalog-model-pangolin","kind":"model","name":"Pangolin","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["dna-genomes"],"method_types":["specialist"]},"source_ids":["catalog-source-pangolin"],"links":[{"relation":"family","target_id":"discovery-model-pangolin"}],"attributes":{"entity_level":"family","version":"published checkpoints","reported_name":"Pangolin","access":"Public specialist code and models under GPL-3.0.","method_type":"specialist","historical_missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"Pangolin predicts splice-site strength and changes caused by genetic variants.","summary_source_ids":["evidence-official-eaf3efab8850672f44e4","evidence-official-de1ee43bdd6f04de9850"],"summary_source_locator":"Paper: Deep neural network architecture and Generating training and test sets; README.md: Usage","sections":[{"title":"How it works","body":"Pangolin predicts splice-site strength and changes caused by genetic variants. Dilated convolutional network with 16 residual blocks and skip connections; separate probability and usage outputs for heart, liver, brain and testis. The documented inputs are VCF or CSV variants, reference FASTA and matching gene annotations; custom sequence inference is also available. The output consists of predicted increases/decreases in splice-site strength and their positions.","source_ids":["evidence-official-eaf3efab8850672f44e4","evidence-official-de1ee43bdd6f04de9850"],"source_locator":"Paper: Deep neural network architecture and Generating training and test sets; README.md: Usage"},{"title":"Versions and reproducibility","body":"Pangolin implementation; gene-annotation database and selected weights must be recorded with a run. 5,000 bases upstream and downstream each output position; minimum 10,001-base input for one prediction, with 15,000-base training blocks producing 5,000 central outputs.","source_ids":["evidence-official-eaf3efab8850672f44e4","evidence-official-de1ee43bdd6f04de9850"],"source_locator":"Paper: Deep neural network architecture and Generating training and test sets; README.md: Usage"}],"facts":[{"label":"Model type","value":"Dilated convolutional splicing predictor","status":"source_checked","source_ids":["evidence-official-eaf3efab8850672f44e4","evidence-official-de1ee43bdd6f04de9850"],"source_locator":"Paper: Deep neural network architecture and Generating training and test sets; README.md: Usage"},{"label":"Architecture","value":"Dilated convolutional network with 16 residual blocks and skip connections; separate probability and usage outputs for heart, liver, brain and testis.","status":"source_checked","source_ids":["evidence-official-eaf3efab8850672f44e4","evidence-official-de1ee43bdd6f04de9850"],"source_locator":"Paper: Deep neural network architecture and Generating training and test sets; README.md: Usage"},{"label":"Inputs","value":"VCF or CSV variants, reference FASTA and matching gene annotations; custom sequence inference is also available.","status":"source_checked","source_ids":["evidence-official-eaf3efab8850672f44e4","evidence-official-de1ee43bdd6f04de9850"],"source_locator":"Paper: Deep neural network architecture and Generating training and test sets; README.md: Usage"},{"label":"Outputs","value":"Predicted increases/decreases in splice-site strength and their positions.","status":"source_checked","source_ids":["evidence-official-eaf3efab8850672f44e4","evidence-official-de1ee43bdd6f04de9850"],"source_locator":"Paper: Deep neural network architecture and Generating training and test sets; README.md: Usage"},{"label":"Parameters","value":"The reviewed architecture section specifies the dilated residual network, but does not give a complete parameter total for the released ensemble.","status":"unreported","source_ids":["evidence-official-eaf3efab8850672f44e4","evidence-official-de1ee43bdd6f04de9850"],"source_locator":"Paper: Deep neural network architecture and Generating training and test sets; README.md: Usage"},{"label":"Known versions","value":"Pangolin implementation; gene-annotation database and selected weights must be recorded with a run.","status":"source_checked","source_ids":["evidence-official-eaf3efab8850672f44e4","evidence-official-de1ee43bdd6f04de9850"],"source_locator":"Paper: Deep neural network architecture and Generating training and test sets; README.md: Usage"},{"label":"Training data","value":"Human, rhesus macaque, mouse and rat sequence/splicing data. Human test chromosomes 1, 3, 5, 7 and 9 are held out, with homologous training genes filtered using Ensembl BioMart.","status":"source_checked","source_ids":["evidence-official-eaf3efab8850672f44e4","evidence-official-de1ee43bdd6f04de9850"],"source_locator":"Paper: Deep neural network architecture and Generating training and test sets; README.md: Usage"},{"label":"Training cutoff","value":"Training annotations are GENCODE 34 (human), Ensembl 100 (rhesus), GENCODE M25 (mouse) and Ensembl 101 (rat). These component releases do not establish one latest RNA-seq collection date.","status":"source_checked","source_ids":["evidence-official-eaf3efab8850672f44e4","evidence-official-de1ee43bdd6f04de9850"],"source_locator":"Paper: Deep neural network architecture and Generating training and test sets; README.md: Usage"},{"label":"Context limits","value":"5,000 bases upstream and downstream each output position; minimum 10,001-base input for one prediction, with 15,000-base training blocks producing 5,000 central outputs.","status":"source_checked","source_ids":["evidence-official-eaf3efab8850672f44e4","evidence-official-de1ee43bdd6f04de9850"],"source_locator":"Paper: Deep neural network architecture and Generating training and test sets; README.md: Usage"},{"label":"Weights licence","value":"Separate checkpoint-distribution terms are not stated in the inspected release documentation and licence material. The source-code licence alone is not recorded as an explicit weight grant.","status":"unreported","source_ids":["evidence-official-eaf3efab8850672f44e4","evidence-official-de1ee43bdd6f04de9850","evidence-official-dd29c6cbb629171059a6"],"source_locator":"Paper: Deep neural network architecture and Generating training and test sets; README.md: Usage; LICENSE: licence text"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/tkzeng/Pangolin","status":"source_checked","source_ids":["evidence-official-eaf3efab8850672f44e4","evidence-official-de1ee43bdd6f04de9850"],"source_locator":"Paper: Deep neural network architecture and Generating training and test sets; README.md: Usage"},{"label":"Code licence","value":"GPL-3.0; inspect the pinned licence and any file-specific terms.","status":"source_checked","source_ids":["evidence-official-dd29c6cbb629171059a6"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"Supports custom sequences and annotation-aware variant scoring with configurable search distance.","source_ids":["evidence-official-eaf3efab8850672f44e4","evidence-official-de1ee43bdd6f04de9850"],"source_locator":"Paper: Deep neural network architecture and Generating training and test sets; README.md: Usage"}],"limitations":[{"text":"Only substitutions and simple insertions/deletions are supported. The documented tool skips variants outside annotated genes, near chromosome ends, inconsistent with the reference or beyond supported deletion lengths.","source_ids":["evidence-official-eaf3efab8850672f44e4","evidence-official-de1ee43bdd6f04de9850"],"source_locator":"Paper: Deep neural network architecture and Generating training and test sets; README.md: Usage"}],"diagram":{"title":"Pangolin workflow","steps":["Variant and reference genome","Construct sequence inputs","Splice-strength prediction","Reference/alternate comparison"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-eaf3efab8850672f44e4","evidence-official-de1ee43bdd6f04de9850"],"source_locator":"Paper: Deep neural network architecture and Generating training and test sets; README.md: Usage"},"coverage":"limited","gaps":["Parameters: The reviewed architecture section specifies the dilated residual network, but does not give a complete parameter total for the released ensemble.","Weights licence: Separate checkpoint-distribution terms are not stated in the inspected release documentation and licence material. The source-code licence alone is not recorded as an explicit weight grant."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}}} {"id":"catalog-model-prokbert","kind":"model","name":"ProkBERT","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["microbes-communities"],"method_types":["foundation model"]},"source_ids":["catalog-source-prokbert"],"links":[],"attributes":{"entity_level":"family","version":"mini","reported_name":"ProkBERT","access":"Public model family and mini checkpoint.","method_type":"foundation model","historical_missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"ProkBERT is a family of microbial DNA encoders used for sequence representation, promoter prediction and phage identification.","summary_source_ids":["evidence-official-2efbfaba5a1c09f4f7fa","evidence-official-965600234c25fbd104c8","evidence-official-ee6e270af125ddf16eb3","evidence-official-5e268f31f0347fc5564c"],"summary_source_locator":"ProkBERT paper Section 2.1.2 and Table 1; official ProkBERT-mini card: Model Description and licence metadata","sections":[{"title":"How it works","body":"ProkBERT is a family of microbial DNA encoders used for sequence representation, promoter prediction and phage identification. MegatronBERT-based encoder using local-context-aware k-mer tokenization, learned relative key/value positions and masked-language pretraining. The documented inputs are microbial DNA segments tokenized with the selected variant. The output consists of sequence representations or predictions from separately fine-tuned promoter/phage heads.","source_ids":["evidence-official-2efbfaba5a1c09f4f7fa","evidence-official-965600234c25fbd104c8","evidence-official-ee6e270af125ddf16eb3","evidence-official-5e268f31f0347fc5564c"],"source_locator":"ProkBERT paper Section 2.1.2 and Table 1; official ProkBERT-mini card: Model Description and licence metadata"},{"title":"Versions and reproducibility","body":"ProkBERT-mini (6-mer, shift 1), mini-c (single base), mini-long (6-mer, shift 2), plus promoter/phage fine-tunes. The paper reports maximum sequence lengths of 1,024bp for mini and 2,048bp for mini-long; tokenization stride distinguishes the variants.","source_ids":["evidence-official-2efbfaba5a1c09f4f7fa","evidence-official-965600234c25fbd104c8","evidence-official-ee6e270af125ddf16eb3","evidence-official-5e268f31f0347fc5564c"],"source_locator":"ProkBERT paper Section 2.1.2 and Table 1; official ProkBERT-mini card: Model Description and licence metadata"}],"facts":[{"label":"Model type","value":"Microbial DNA BERT encoder","status":"source_checked","source_ids":["evidence-official-2efbfaba5a1c09f4f7fa","evidence-official-965600234c25fbd104c8","evidence-official-ee6e270af125ddf16eb3","evidence-official-5e268f31f0347fc5564c"],"source_locator":"ProkBERT paper Section 2.1.2 and Table 1; official ProkBERT-mini card: Model Description and licence metadata"},{"label":"Architecture","value":"MegatronBERT-based encoder using local-context-aware k-mer tokenization, learned relative key/value positions and masked-language pretraining.","status":"source_checked","source_ids":["evidence-official-2efbfaba5a1c09f4f7fa","evidence-official-965600234c25fbd104c8","evidence-official-ee6e270af125ddf16eb3","evidence-official-5e268f31f0347fc5564c"],"source_locator":"ProkBERT paper Section 2.1.2 and Table 1; official ProkBERT-mini card: Model Description and licence metadata"},{"label":"Inputs","value":"Microbial DNA segments tokenized with the selected variant.","status":"source_checked","source_ids":["evidence-official-2efbfaba5a1c09f4f7fa","evidence-official-965600234c25fbd104c8","evidence-official-ee6e270af125ddf16eb3","evidence-official-5e268f31f0347fc5564c"],"source_locator":"ProkBERT paper Section 2.1.2 and Table 1; official ProkBERT-mini card: Model Description and licence metadata"},{"label":"Outputs","value":"Sequence representations or predictions from separately fine-tuned promoter/phage heads.","status":"source_checked","source_ids":["evidence-official-2efbfaba5a1c09f4f7fa","evidence-official-965600234c25fbd104c8","evidence-official-ee6e270af125ddf16eb3","evidence-official-5e268f31f0347fc5564c"],"source_locator":"ProkBERT paper Section 2.1.2 and Table 1; official ProkBERT-mini card: Model Description and licence metadata"},{"label":"Parameters","value":"20.6M for the inspected ProkBERT-mini checkpoint; do not assign that count automatically to all variants.","status":"source_checked","source_ids":["evidence-official-2efbfaba5a1c09f4f7fa","evidence-official-965600234c25fbd104c8","evidence-official-ee6e270af125ddf16eb3","evidence-official-5e268f31f0347fc5564c"],"source_locator":"ProkBERT paper Section 2.1.2 and Table 1; official ProkBERT-mini card: Model Description and licence metadata"},{"label":"Known versions","value":"ProkBERT-mini (6-mer, shift 1), mini-c (single base), mini-long (6-mer, shift 2), plus promoter/phage fine-tunes.","status":"source_checked","source_ids":["evidence-official-2efbfaba5a1c09f4f7fa","evidence-official-965600234c25fbd104c8","evidence-official-ee6e270af125ddf16eb3","evidence-official-5e268f31f0347fc5564c"],"source_locator":"ProkBERT paper Section 2.1.2 and Table 1; official ProkBERT-mini card: Model Description and licence metadata"},{"label":"Training data","value":"NCBI RefSeq genomes covering bacteria, viruses, archaea and fungi; README reports 976,878 contigs from 17,178 assemblies and 3,882 genera.","status":"source_checked","source_ids":["evidence-official-2efbfaba5a1c09f4f7fa","evidence-official-965600234c25fbd104c8","evidence-official-ee6e270af125ddf16eb3","evidence-official-5e268f31f0347fc5564c"],"source_locator":"ProkBERT paper Section 2.1.2 and Table 1; official ProkBERT-mini card: Model Description and licence metadata"},{"label":"Training cutoff","value":"NCBI RefSeq training-genome retrieval date: 6 January 2023, as reported in Methods.","status":"source_checked","source_ids":["evidence-official-2efbfaba5a1c09f4f7fa","evidence-official-965600234c25fbd104c8","evidence-official-ee6e270af125ddf16eb3","evidence-official-5e268f31f0347fc5564c"],"source_locator":"ProkBERT paper Section 2.1.2 and Table 1; official ProkBERT-mini card: Model Description and licence metadata"},{"label":"Context limits","value":"The paper reports maximum sequence lengths of 1,024bp for mini and 2,048bp for mini-long; tokenization stride distinguishes the variants.","status":"source_checked","source_ids":["evidence-official-2efbfaba5a1c09f4f7fa","evidence-official-965600234c25fbd104c8","evidence-official-ee6e270af125ddf16eb3","evidence-official-5e268f31f0347fc5564c"],"source_locator":"ProkBERT paper Section 2.1.2 and Table 1; official ProkBERT-mini card: Model Description and licence metadata"},{"label":"Weights licence","value":"The official neuralbioinfo/prokbert-mini card declares CC-BY-NC-4.0. The code repository is MIT, so code and model-weight rights differ.","status":"source_checked","source_ids":["evidence-official-2efbfaba5a1c09f4f7fa","evidence-official-965600234c25fbd104c8","evidence-official-ee6e270af125ddf16eb3","evidence-official-5e268f31f0347fc5564c"],"source_locator":"ProkBERT paper Section 2.1.2 and Table 1; official ProkBERT-mini card: Model Description and licence metadata"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/nbrg-ppcu/prokbert","status":"source_checked","source_ids":["evidence-official-2efbfaba5a1c09f4f7fa","evidence-official-965600234c25fbd104c8","evidence-official-ee6e270af125ddf16eb3","evidence-official-5e268f31f0347fc5564c"],"source_locator":"ProkBERT paper Section 2.1.2 and Table 1; official ProkBERT-mini card: Model Description and licence metadata"},{"label":"Code licence","value":"MIT","status":"source_checked","source_ids":["evidence-official-03106594cde7719aa359"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"The released family includes character and shifted k-mer variants plus task-specific fine-tuned models.","source_ids":["evidence-official-2efbfaba5a1c09f4f7fa","evidence-official-965600234c25fbd104c8","evidence-official-ee6e270af125ddf16eb3","evidence-official-5e268f31f0347fc5564c"],"source_locator":"ProkBERT paper Section 2.1.2 and Table 1; official ProkBERT-mini card: Model Description and licence metadata"}],"limitations":[{"text":"The authors explicitly identify restricted context as a limitation in phage analysis. A promoter or phage classifier is a distinct fine-tuned configuration.","source_ids":["evidence-official-2efbfaba5a1c09f4f7fa","evidence-official-965600234c25fbd104c8","evidence-official-ee6e270af125ddf16eb3","evidence-official-5e268f31f0347fc5564c"],"source_locator":"ProkBERT paper Section 2.1.2 and Table 1; official ProkBERT-mini card: Model Description and licence metadata"}],"diagram":{"title":"ProkBERT workflow","steps":["DNA segment","Variant-specific LCA tokens","BERT encoder","Representation or task head"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-2efbfaba5a1c09f4f7fa","evidence-official-965600234c25fbd104c8","evidence-official-ee6e270af125ddf16eb3","evidence-official-5e268f31f0347fc5564c"],"source_locator":"ProkBERT paper Section 2.1.2 and Table 1; official ProkBERT-mini card: Model Description and licence metadata"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}}} {"id":"catalog-model-proteinmpnn","kind":"model","name":"ProteinMPNN","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["proteins-complexes"],"method_types":["specialist"]},"source_ids":["catalog-source-proteinmpnn"],"links":[{"relation":"variant_of","target_id":"discovery-model-proteinmpnn"}],"attributes":{"entity_level":"family","version":"v_48_020","reported_name":"ProteinMPNN","access":"Public code and checkpoints; requires a suitable protein structure.","method_type":"specialist","historical_missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"ProteinMPNN designs amino-acid sequences for a supplied protein backbone.","summary_source_ids":["evidence-official-ea568a52ea3ab476db6a","evidence-official-7b6fdf915d9c6950ad0d","evidence-official-019aff235ae2c3cf29d6","evidence-official-92b0ae57a020cbc40fe7","evidence-official-0b8003691827eb09b0a4"],"summary_source_locator":"ProteinMPNN paper: main-text model development; training/README.md: multi-chain training set, list.csv and cluster manifests; protein_mpnn_utils.py: ProteinMPNN; Supplementary Materials: Methods for training multi-chain models / Training data","sections":[{"title":"How it works","body":"ProteinMPNN converts a supplied backbone into a graph whose edges encode interatomic distances. Message-passing layers update node and edge features, and an autoregressive decoder samples amino acids while conditioning on the backbone and previously assigned residues. Fixed residues, tied positions and chain choices change the design task and must accompany its result.","source_ids":["evidence-official-ea568a52ea3ab476db6a","evidence-official-7b6fdf915d9c6950ad0d","evidence-official-019aff235ae2c3cf29d6","evidence-official-92b0ae57a020cbc40fe7","evidence-official-0b8003691827eb09b0a4"],"source_locator":"ProteinMPNN paper: main-text model development; training/README.md: multi-chain training set, list.csv and cluster manifests; protein_mpnn_utils.py: ProteinMPNN; Supplementary Materials: Methods for training multi-chain models / Training data"},{"title":"Versions and reproducibility","body":"v_48_002, v_48_010, v_48_020 and v_48_030; distinct soluble and C-alpha-only weights. Structure-size and memory dependent. README --max_length is an implementation guard, not a validated scientific context limit.","source_ids":["evidence-official-ea568a52ea3ab476db6a","evidence-official-7b6fdf915d9c6950ad0d","evidence-official-019aff235ae2c3cf29d6","evidence-official-92b0ae57a020cbc40fe7","evidence-official-0b8003691827eb09b0a4"],"source_locator":"ProteinMPNN paper: main-text model development; training/README.md: multi-chain training set, list.csv and cluster manifests; protein_mpnn_utils.py: ProteinMPNN; Supplementary Materials: Methods for training multi-chain models / Training data"}],"facts":[{"label":"Model type","value":"Structure-conditioned message-passing sequence design model","status":"source_checked","source_ids":["evidence-official-ea568a52ea3ab476db6a","evidence-official-7b6fdf915d9c6950ad0d","evidence-official-019aff235ae2c3cf29d6","evidence-official-92b0ae57a020cbc40fe7","evidence-official-0b8003691827eb09b0a4"],"source_locator":"ProteinMPNN paper: main-text model development; training/README.md: multi-chain training set, list.csv and cluster manifests; protein_mpnn_utils.py: ProteinMPNN; Supplementary Materials: Methods for training multi-chain models / Training data"},{"label":"Architecture","value":"Message-passing encoder-decoder with structural interatomic-distance features and edge updates; sequences are sampled with the configured autoregressive decoding procedure.","status":"source_checked","source_ids":["evidence-official-ea568a52ea3ab476db6a","evidence-official-7b6fdf915d9c6950ad0d","evidence-official-019aff235ae2c3cf29d6","evidence-official-92b0ae57a020cbc40fe7","evidence-official-0b8003691827eb09b0a4"],"source_locator":"ProteinMPNN paper: main-text model development; training/README.md: multi-chain training set, list.csv and cluster manifests; protein_mpnn_utils.py: ProteinMPNN; Supplementary Materials: Methods for training multi-chain models / Training data"},{"label":"Inputs","value":"Protein backbone coordinates, with optional fixed residues, chain choices, tied positions and amino-acid constraints.","status":"source_checked","source_ids":["evidence-official-ea568a52ea3ab476db6a","evidence-official-7b6fdf915d9c6950ad0d","evidence-official-019aff235ae2c3cf29d6","evidence-official-92b0ae57a020cbc40fe7","evidence-official-0b8003691827eb09b0a4"],"source_locator":"ProteinMPNN paper: main-text model development; training/README.md: multi-chain training set, list.csv and cluster manifests; protein_mpnn_utils.py: ProteinMPNN; Supplementary Materials: Methods for training multi-chain models / Training data"},{"label":"Outputs","value":"Designed sequences, sequence scores and conditional amino-acid probabilities.","status":"source_checked","source_ids":["evidence-official-ea568a52ea3ab476db6a","evidence-official-7b6fdf915d9c6950ad0d","evidence-official-019aff235ae2c3cf29d6","evidence-official-92b0ae57a020cbc40fe7","evidence-official-0b8003691827eb09b0a4"],"source_locator":"ProteinMPNN paper: main-text model development; training/README.md: multi-chain training set, list.csv and cluster manifests; protein_mpnn_utils.py: ProteinMPNN; Supplementary Materials: Methods for training multi-chain models / Training data"},{"label":"Parameters","value":"The inspected model implementation is configured through encoder/decoder depth and feature width. The paper and training README do not state an exact total for every released checkpoint.","status":"unreported","source_ids":["evidence-official-ea568a52ea3ab476db6a","evidence-official-7b6fdf915d9c6950ad0d","evidence-official-019aff235ae2c3cf29d6","evidence-official-92b0ae57a020cbc40fe7","evidence-official-0b8003691827eb09b0a4"],"source_locator":"ProteinMPNN paper: main-text model development; training/README.md: multi-chain training set, list.csv and cluster manifests; protein_mpnn_utils.py: ProteinMPNN; Supplementary Materials: Methods for training multi-chain models / Training data"},{"label":"Known versions","value":"v_48_002, v_48_010, v_48_020 and v_48_030; distinct soluble and C-alpha-only weights.","status":"source_checked","source_ids":["evidence-official-ea568a52ea3ab476db6a","evidence-official-7b6fdf915d9c6950ad0d","evidence-official-019aff235ae2c3cf29d6","evidence-official-92b0ae57a020cbc40fe7","evidence-official-0b8003691827eb09b0a4"],"source_locator":"ProteinMPNN paper: main-text model development; training/README.md: multi-chain training set, list.csv and cluster manifests; protein_mpnn_utils.py: ProteinMPNN; Supplementary Materials: Methods for training multi-chain models / Training data"},{"label":"Training data","value":"Released multi-chain training set of PDB biological units, with chain metadata and validation/test cluster manifests. The documented set is dated 2 August 2021.","status":"source_checked","source_ids":["evidence-official-ea568a52ea3ab476db6a","evidence-official-7b6fdf915d9c6950ad0d","evidence-official-019aff235ae2c3cf29d6","evidence-official-92b0ae57a020cbc40fe7","evidence-official-0b8003691827eb09b0a4"],"source_locator":"ProteinMPNN paper: main-text model development; training/README.md: multi-chain training set, list.csv and cluster manifests; protein_mpnn_utils.py: ProteinMPNN; Supplementary Materials: Methods for training multi-chain models / Training data"},{"label":"Training cutoff","value":"The released PDB training-set snapshot is dated 2021-08-02; preserve its chain-level deposition metadata and cluster split for a run.","status":"source_checked","source_ids":["evidence-official-ea568a52ea3ab476db6a","evidence-official-7b6fdf915d9c6950ad0d","evidence-official-019aff235ae2c3cf29d6","evidence-official-92b0ae57a020cbc40fe7","evidence-official-0b8003691827eb09b0a4"],"source_locator":"ProteinMPNN paper: main-text model development; training/README.md: multi-chain training set, list.csv and cluster manifests; protein_mpnn_utils.py: ProteinMPNN; Supplementary Materials: Methods for training multi-chain models / Training data"},{"label":"Context limits","value":"Structure-size and memory dependent. README --max_length is an implementation guard, not a validated scientific context limit.","status":"source_checked","source_ids":["evidence-official-ea568a52ea3ab476db6a","evidence-official-7b6fdf915d9c6950ad0d","evidence-official-019aff235ae2c3cf29d6","evidence-official-92b0ae57a020cbc40fe7","evidence-official-0b8003691827eb09b0a4"],"source_locator":"ProteinMPNN paper: main-text model development; training/README.md: multi-chain training set, list.csv and cluster manifests; protein_mpnn_utils.py: ProteinMPNN; Supplementary Materials: Methods for training multi-chain models / Training data"},{"label":"Weights licence","value":"Separate checkpoint-distribution terms are not stated in the inspected release documentation and licence material. The source-code licence alone is not recorded as an explicit weight grant.","status":"unreported","source_ids":["evidence-official-ea568a52ea3ab476db6a","evidence-official-7b6fdf915d9c6950ad0d","evidence-official-019aff235ae2c3cf29d6","evidence-official-92b0ae57a020cbc40fe7","evidence-official-0b8003691827eb09b0a4","evidence-official-eebe7c91156963e6ddc0"],"source_locator":"ProteinMPNN paper: main-text model development; training/README.md: multi-chain training set, list.csv and cluster manifests; protein_mpnn_utils.py: ProteinMPNN; Supplementary Materials: Methods for training multi-chain models / Training data; LICENSE: licence text"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/dauparas/ProteinMPNN","status":"source_checked","source_ids":["evidence-official-ea568a52ea3ab476db6a","evidence-official-7b6fdf915d9c6950ad0d","evidence-official-019aff235ae2c3cf29d6","evidence-official-92b0ae57a020cbc40fe7","evidence-official-0b8003691827eb09b0a4"],"source_locator":"ProteinMPNN paper: main-text model development; training/README.md: multi-chain training set, list.csv and cluster manifests; protein_mpnn_utils.py: ProteinMPNN; Supplementary Materials: Methods for training multi-chain models / Training data"},{"label":"Code licence","value":"MIT","status":"source_checked","source_ids":["evidence-official-eebe7c91156963e6ddc0"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"The interface makes design constraints explicit and includes full-backbone, C-alpha-only and soluble-protein weight sets.","source_ids":["evidence-official-ea568a52ea3ab476db6a","evidence-official-7b6fdf915d9c6950ad0d","evidence-official-019aff235ae2c3cf29d6","evidence-official-92b0ae57a020cbc40fe7","evidence-official-0b8003691827eb09b0a4"],"source_locator":"ProteinMPNN paper: main-text model development; training/README.md: multi-chain training set, list.csv and cluster manifests; protein_mpnn_utils.py: ProteinMPNN; Supplementary Materials: Methods for training multi-chain models / Training data"}],"limitations":[{"text":"The requested backbone and constraints are part of the evaluated problem. A command-line maximum-length guard is not evidence that designs at that size have been validated.","source_ids":["evidence-official-ea568a52ea3ab476db6a","evidence-official-7b6fdf915d9c6950ad0d","evidence-official-019aff235ae2c3cf29d6","evidence-official-92b0ae57a020cbc40fe7","evidence-official-0b8003691827eb09b0a4"],"source_locator":"ProteinMPNN paper: main-text model development; training/README.md: multi-chain training set, list.csv and cluster manifests; protein_mpnn_utils.py: ProteinMPNN; Supplementary Materials: Methods for training multi-chain models / Training data"}],"diagram":{"title":"ProteinMPNN workflow","steps":["Protein backbone","Structural graph features","Message-passing model","Constrained sequence sampling"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-ea568a52ea3ab476db6a","evidence-official-7b6fdf915d9c6950ad0d","evidence-official-019aff235ae2c3cf29d6","evidence-official-92b0ae57a020cbc40fe7","evidence-official-0b8003691827eb09b0a4"],"source_locator":"ProteinMPNN paper: main-text model development; training/README.md: multi-chain training set, list.csv and cluster manifests; protein_mpnn_utils.py: ProteinMPNN; Supplementary Materials: Methods for training multi-chain models / Training data"},"coverage":"limited","gaps":["Parameters: The inspected model implementation is configured through encoder/decoder depth and feature width. The paper and training README do not state an exact total for every released checkpoint.","Weights licence: Separate checkpoint-distribution terms are not stated in the inspected release documentation and licence material. The source-code licence alone is not recorded as an explicit weight grant."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}}} {"id":"catalog-model-rhofold","kind":"model","name":"RhoFold+","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["rna-transcriptomes"],"method_types":["specialist"]},"source_ids":["catalog-source-rhofold"],"links":[],"attributes":{"entity_level":"family","version":"pretrained","reported_name":"RhoFold+","access":"Public code and checkpoint instructions.","method_type":"specialist","historical_missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"RhoFold+ predicts RNA three-dimensional structures from RNA sequence with alignment information.","summary_source_ids":["evidence-official-955922130ff85a74fa18","evidence-official-8b4e1bd487c21cda9862","evidence-official-3e1890c3652bc5c150f7"],"summary_source_locator":"Paper: Automated end-to-end platform, Large-scale pretraining dataset, Efficient development of a self-distillation dataset, Feature processing with Rhoformer and Data availability; README.md: Usage; official cuhkaih/rhofold card licence metadata","sections":[{"title":"How it works","body":"RhoFold+ predicts RNA three-dimensional structures from RNA sequence with alignment information. RNA-FM embeddings and MSA representations enter the Rhoformer transformer stack; a geometry-aware invariant-point-attention structure module predicts frames and torsion angles with recycling. The documented inputs are RNA FASTA (A, U, G, C), optionally a supplied MSA; an MSA is otherwise generated by the full workflow. The output consists of three-dimensional PDB models, predicted distograms, secondary structure and per-residue confidence.","source_ids":["evidence-official-955922130ff85a74fa18","evidence-official-8b4e1bd487c21cda9862","evidence-official-3e1890c3652bc5c150f7"],"source_locator":"Paper: Automated end-to-end platform, Large-scale pretraining dataset, Efficient development of a self-distillation dataset, Feature processing with Rhoformer and Data availability; README.md: Usage; official cuhkaih/rhofold card licence metadata"},{"title":"Versions and reproducibility","body":"RhoFold+; the README links a pretrained checkpoint and the 2024 Nature Methods paper. The paper limits MSA depth to256 sequences during training and default inference. This is alignment depth, not an RNA-length maximum; the latter remains unextracted.","source_ids":["evidence-official-955922130ff85a74fa18","evidence-official-8b4e1bd487c21cda9862","evidence-official-3e1890c3652bc5c150f7"],"source_locator":"Paper: Automated end-to-end platform, Large-scale pretraining dataset, Efficient development of a self-distillation dataset, Feature processing with Rhoformer and Data availability; README.md: Usage; official cuhkaih/rhofold card licence metadata"}],"facts":[{"label":"Model type","value":"RNA structure predictor with language-model and MSA inputs","status":"source_checked","source_ids":["evidence-official-955922130ff85a74fa18","evidence-official-8b4e1bd487c21cda9862","evidence-official-3e1890c3652bc5c150f7"],"source_locator":"Paper: Automated end-to-end platform, Large-scale pretraining dataset, Efficient development of a self-distillation dataset, Feature processing with Rhoformer and Data availability; README.md: Usage; official cuhkaih/rhofold card licence metadata"},{"label":"Architecture","value":"RNA-FM embeddings and MSA representations enter the Rhoformer transformer stack; a geometry-aware invariant-point-attention structure module predicts frames and torsion angles with recycling.","status":"source_checked","source_ids":["evidence-official-955922130ff85a74fa18","evidence-official-8b4e1bd487c21cda9862","evidence-official-3e1890c3652bc5c150f7"],"source_locator":"Paper: Automated end-to-end platform, Large-scale pretraining dataset, Efficient development of a self-distillation dataset, Feature processing with Rhoformer and Data availability; README.md: Usage; official cuhkaih/rhofold card licence metadata"},{"label":"Inputs","value":"RNA FASTA (A, U, G, C), optionally a supplied MSA; an MSA is otherwise generated by the full workflow.","status":"source_checked","source_ids":["evidence-official-955922130ff85a74fa18","evidence-official-8b4e1bd487c21cda9862","evidence-official-3e1890c3652bc5c150f7"],"source_locator":"Paper: Automated end-to-end platform, Large-scale pretraining dataset, Efficient development of a self-distillation dataset, Feature processing with Rhoformer and Data availability; README.md: Usage; official cuhkaih/rhofold card licence metadata"},{"label":"Outputs","value":"Three-dimensional PDB models, predicted distograms, secondary structure and per-residue confidence.","status":"source_checked","source_ids":["evidence-official-955922130ff85a74fa18","evidence-official-8b4e1bd487c21cda9862","evidence-official-3e1890c3652bc5c150f7"],"source_locator":"Paper: Automated end-to-end platform, Large-scale pretraining dataset, Efficient development of a self-distillation dataset, Feature processing with Rhoformer and Data availability; README.md: Usage; official cuhkaih/rhofold card licence metadata"},{"label":"Parameters","value":"The inspected paper describes RNA-FM, Rhoformer and structure modules without stating a total for the complete selected predictor in those architecture sections.","status":"unreported","source_ids":["evidence-official-955922130ff85a74fa18","evidence-official-8b4e1bd487c21cda9862","evidence-official-3e1890c3652bc5c150f7"],"source_locator":"Paper: Automated end-to-end platform, Large-scale pretraining dataset, Efficient development of a self-distillation dataset, Feature processing with Rhoformer and Data availability; README.md: Usage; official cuhkaih/rhofold card licence metadata"},{"label":"Known versions","value":"RhoFold+; the README links a pretrained checkpoint and the 2024 Nature Methods paper.","status":"source_checked","source_ids":["evidence-official-955922130ff85a74fa18","evidence-official-8b4e1bd487c21cda9862","evidence-official-3e1890c3652bc5c150f7"],"source_locator":"Paper: Automated end-to-end platform, Large-scale pretraining dataset, Efficient development of a self-distillation dataset, Feature processing with Rhoformer and Data availability; README.md: Usage; official cuhkaih/rhofold card licence metadata"},{"label":"Training data","value":"RNA-FM pretraining uses RNAcentral100. Structure training uses PDB RNA chains selected through BGSU representative sets, with additional self-distillation from RNAStralign/bpRNA-derived sequences.","status":"source_checked","source_ids":["evidence-official-955922130ff85a74fa18","evidence-official-8b4e1bd487c21cda9862","evidence-official-3e1890c3652bc5c150f7"],"source_locator":"Paper: Automated end-to-end platform, Large-scale pretraining dataset, Efficient development of a self-distillation dataset, Feature processing with Rhoformer and Data availability; README.md: Usage; official cuhkaih/rhofold card licence metadata"},{"label":"Training cutoff","value":"Structural-data selection uses the BGSU representative set dated 2022-04-13. This does not establish a single cutoff for every RNA-FM or MSA resource.","status":"source_checked","source_ids":["evidence-official-955922130ff85a74fa18","evidence-official-8b4e1bd487c21cda9862","evidence-official-3e1890c3652bc5c150f7"],"source_locator":"Paper: Automated end-to-end platform, Large-scale pretraining dataset, Efficient development of a self-distillation dataset, Feature processing with Rhoformer and Data availability; README.md: Usage; official cuhkaih/rhofold card licence metadata"},{"label":"Context limits","value":"MSA depth is capped at 256 during documented training and default inference. The reviewed paper does not establish a single RNA-length maximum; MSA depth is a different quantity.","status":"unreported","source_ids":["evidence-official-955922130ff85a74fa18","evidence-official-8b4e1bd487c21cda9862","evidence-official-3e1890c3652bc5c150f7"],"source_locator":"Paper: Automated end-to-end platform, Large-scale pretraining dataset, Efficient development of a self-distillation dataset, Feature processing with Rhoformer and Data availability; README.md: Usage; official cuhkaih/rhofold card licence metadata"},{"label":"Weights licence","value":"Apache-2.0 declared in the official cuhkaih/rhofold model-card metadata; this is distinct from access conditions for training data.","status":"source_checked","source_ids":["evidence-official-955922130ff85a74fa18","evidence-official-8b4e1bd487c21cda9862","evidence-official-3e1890c3652bc5c150f7"],"source_locator":"Paper: Automated end-to-end platform, Large-scale pretraining dataset, Efficient development of a self-distillation dataset, Feature processing with Rhoformer and Data availability; README.md: Usage; official cuhkaih/rhofold card licence metadata"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/ml4bio/RhoFold","status":"source_checked","source_ids":["evidence-official-955922130ff85a74fa18","evidence-official-8b4e1bd487c21cda9862","evidence-official-3e1890c3652bc5c150f7"],"source_locator":"Paper: Automated end-to-end platform, Large-scale pretraining dataset, Efficient development of a self-distillation dataset, Feature processing with Rhoformer and Data availability; README.md: Usage; official cuhkaih/rhofold card licence metadata"},{"label":"Code licence","value":"Apache-2.0","status":"source_checked","source_ids":["evidence-official-4e0280294c1215534077"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"Produces secondary and tertiary outputs and supports provided or automatically generated alignments.","source_ids":["evidence-official-955922130ff85a74fa18","evidence-official-8b4e1bd487c21cda9862","evidence-official-3e1890c3652bc5c150f7"],"source_locator":"Paper: Automated end-to-end platform, Large-scale pretraining dataset, Efficient development of a self-distillation dataset, Feature processing with Rhoformer and Data availability; README.md: Usage; official cuhkaih/rhofold card licence metadata"}],"limitations":[{"text":"The README labels sequence-only inference as a lower-accuracy testing mode. Constructing the full MSA databases needs substantial local storage; macOS is not supported by its documented setup.","source_ids":["evidence-official-955922130ff85a74fa18","evidence-official-8b4e1bd487c21cda9862","evidence-official-3e1890c3652bc5c150f7"],"source_locator":"Paper: Automated end-to-end platform, Large-scale pretraining dataset, Efficient development of a self-distillation dataset, Feature processing with Rhoformer and Data availability; README.md: Usage; official cuhkaih/rhofold card licence metadata"}],"diagram":{"title":"RhoFold+ workflow","steps":["RNA sequence and MSA","RhoFold+ prediction","Structure and confidence","Optional relaxation"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-955922130ff85a74fa18","evidence-official-8b4e1bd487c21cda9862","evidence-official-3e1890c3652bc5c150f7"],"source_locator":"Paper: Automated end-to-end platform, Large-scale pretraining dataset, Efficient development of a self-distillation dataset, Feature processing with Rhoformer and Data availability; README.md: Usage; official cuhkaih/rhofold card licence metadata"},"coverage":"limited","gaps":["Parameters: The inspected paper describes RNA-FM, Rhoformer and structure modules without stating a total for the complete selected predictor in those architecture sections.","Context limits: MSA depth is capped at 256 during documented training and default inference. The reviewed paper does not establish a single RNA-length maximum; MSA depth is a different quantity."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}}} {"id":"catalog-model-rna-fm","kind":"model","name":"RNA-FM","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["rna-transcriptomes"],"method_types":["foundation model"]},"source_ids":["catalog-source-rna-fm"],"links":[{"relation":"family","target_id":"discovery-model-rna-fm"}],"attributes":{"entity_level":"family","version":"ncRNA","reported_name":"RNA-FM","access":"Public code and checkpoint instructions.","method_type":"foundation model","historical_missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"RNA-FM learns contextual representations of RNA nucleotides for downstream RNA analyses.","summary_source_ids":["evidence-official-fd8e332abdf04a75195b","evidence-official-e17e3864c464d981afc7"],"summary_source_locator":"README.md: Quick Start, embedding examples and RNA foundation model comparison table","sections":[{"title":"How it works","body":"RNA-FM converts an RNA sequence into one token per nucleotide. Twelve transformer encoder blocks use self-attention to produce a 640-dimensional representation at each position. During pretraining, the model learns to recover masked nucleotides from their surrounding sequence. The resulting representations can be supplied to a separately specified downstream model; they are not, by themselves, a structure or functional prediction.","source_ids":["evidence-final-model-rna-fm-paper"],"source_locator":"arXiv:2204.00300v5, Methods: ncRNA data collection and preprocessing and RNA foundation model training details (p.22)"},{"title":"Versions and reproducibility","body":"The nucleotide-based rna_fm_t12 and codon-based mrna_fm_t12 interfaces are distinct. The original RNA-FM paper sets a training input-length limit of 1,024 and describes a usable limit of 1,022 nucleotides. That limit should not be assigned to mRNA-FM without checking its separate configuration.","source_ids":["evidence-official-fd8e332abdf04a75195b","evidence-official-e17e3864c464d981afc7","evidence-final-model-rna-fm-paper"],"source_locator":"Official README: Quick Start and RNA-FM/mRNA-FM examples; arXiv:2204.00300v5, Methods: training input length and SARS-CoV-2 genome embedding extraction (pp.22–23)"}],"facts":[{"label":"Model type","value":"Masked-token RNA transformer encoder","status":"source_checked","source_ids":["evidence-official-fd8e332abdf04a75195b","evidence-official-e17e3864c464d981afc7"],"source_locator":"README.md: Quick Start, embedding examples and RNA foundation model comparison table"},{"label":"Architecture","value":"12-layer masked-token transformer encoder with hidden width 640 and 20 attention heads; nucleotide tokens produce contextual representations.","status":"source_checked","source_ids":["evidence-final-model-rna-fm-paper"],"source_locator":"arXiv:2204.00300v5, Methods: ncRNA data collection and preprocessing and RNA foundation model training details (p.22)"},{"label":"Inputs","value":"RNA sequences tokenized at nucleotide resolution.","status":"source_checked","source_ids":["evidence-official-fd8e332abdf04a75195b","evidence-official-e17e3864c464d981afc7"],"source_locator":"README.md: Quick Start, embedding examples and RNA foundation model comparison table"},{"label":"Outputs","value":"Contextual token embeddings for a specified downstream RNA task.","status":"source_checked","source_ids":["evidence-official-fd8e332abdf04a75195b","evidence-official-e17e3864c464d981afc7"],"source_locator":"README.md: Quick Start, embedding examples and RNA foundation model comparison table"},{"label":"Parameters","value":"99M, as printed in the official Foundation Models table.","status":"source_checked","source_ids":["evidence-official-fd8e332abdf04a75195b","evidence-official-e17e3864c464d981afc7"],"source_locator":"README.md: Quick Start, embedding examples and RNA foundation model comparison table"},{"label":"Known versions","value":"rna_fm_t12 and mrna_fm_t12 are separate pretrained interfaces.","status":"source_checked","source_ids":["evidence-official-fd8e332abdf04a75195b","evidence-official-e17e3864c464d981afc7"],"source_locator":"README.md: Quick Start, embedding examples and RNA foundation model comparison table"},{"label":"Training data","value":"23.7 million non-coding RNA sequences collected from RNAcentral. The authors replace T with U and remove identical sequences using CD-HIT-EST at 100% identity, naming the resulting corpus RNAcentral100.","status":"source_checked","source_ids":["evidence-final-model-rna-fm-paper"],"source_locator":"arXiv:2204.00300v5, Methods: ncRNA data collection and preprocessing and RNA foundation model training details (p.22)"},{"label":"Training cutoff","value":"The inspected Methods and official README do not establish an exact dated RNAcentral release. RNAcentral100 is the authors’ processed-corpus label, not a verified release number.","status":"unreported","source_ids":["evidence-final-model-rna-fm-paper","evidence-official-fd8e332abdf04a75195b"],"source_locator":"arXiv:2204.00300v5, Methods: ncRNA data collection and preprocessing (p.22); official README: Foundation Models table"},{"label":"Context limits","value":"The original paper sets a training input-length limit of 1,024 and describes a usable input limit of 1,022 nucleotides. These are the original RNA-FM settings, not a validated limit for later codon-based mRNA-FM checkpoints.","status":"source_checked","source_ids":["evidence-final-model-rna-fm-paper"],"source_locator":"arXiv:2204.00300v5, Methods: RNA foundation model training details (pp.22–23) and RNA-FM application input-limit statement (p.23)"},{"label":"Weights licence","value":"Separate checkpoint-distribution terms are not stated in the inspected release documentation and licence material. The source-code licence alone is not recorded as an explicit weight grant.","status":"unreported","source_ids":["evidence-official-fd8e332abdf04a75195b","evidence-official-e17e3864c464d981afc7","evidence-official-3fde3df73e79e455bd86"],"source_locator":"README.md: Quick Start, embedding examples and RNA foundation model comparison table; LICENSE: licence text"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/ml4bio/RNA-FM","status":"source_checked","source_ids":["evidence-official-fd8e332abdf04a75195b","evidence-official-e17e3864c464d981afc7"],"source_locator":"README.md: Quick Start, embedding examples and RNA foundation model comparison table"},{"label":"Code licence","value":"MIT","status":"source_checked","source_ids":["evidence-official-3fde3df73e79e455bd86"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"The repository exposes embedding extraction and examples for downstream RNA analyses.","source_ids":["evidence-official-fd8e332abdf04a75195b","evidence-official-e17e3864c464d981afc7"],"source_locator":"README.md: Quick Start, embedding examples and RNA foundation model comparison table"}],"limitations":[{"text":"Base-level RNA-FM and codon-level mRNA-FM are not interchangeable. The task head and tokenization must be specified in each evaluation.","source_ids":["evidence-official-fd8e332abdf04a75195b","evidence-official-e17e3864c464d981afc7"],"source_locator":"README.md: Quick Start, embedding examples and RNA foundation model comparison table"}],"diagram":{"title":"RNA-FM workflow","steps":["RNA sequence","Nucleotide tokenizer","12-layer transformer encoder","Contextual nucleotide representations"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-fd8e332abdf04a75195b","evidence-official-e17e3864c464d981afc7"],"source_locator":"README.md: Quick Start, embedding examples and RNA foundation model comparison table"},"coverage":"limited","gaps":["Weights licence: Separate checkpoint-distribution terms are not stated in the inspected release documentation and licence material. The source-code licence alone is not recorded as an explicit weight grant.","Training cutoff: An exact dated RNAcentral release is not established by the inspected original Methods or official README."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied. Follow-up retrieved the complete original RNA-FM PDF and inspected the full pinned Geneformer repository inventory; unavailable labels were updated only where new evidence resolved the earlier retrieval gap. Follow-up audit reconciles the narrative with the verified RNA-FM input limit and clarifies the original paper’s RNAcentral100 preprocessing definition; mRNA-FM remains separate."}}}} {"id":"catalog-model-scfoundation","kind":"model","name":"scFoundation","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["cells-tissues"],"method_types":["foundation model"]},"source_ids":["catalog-source-scfoundation"],"links":[],"attributes":{"entity_level":"family","version":"100M","reported_name":"scFoundation","access":"Public code; model weights have separate terms that must be checked.","method_type":"foundation model","historical_missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"scFoundation produces contextual cell and gene representations from gene-expression measurements.","summary_source_ids":["evidence-official-bfbc8babf3630ddb3ed7","evidence-official-7972ce2bdd6a5e40d8ad","evidence-official-9863509ad10b824dc96b"],"summary_source_locator":"June 15 2023 scFoundation manuscript: Results pre-training framework; Methods Data collection, Embedding, Encoder, Decoder and Read-depth-aware pre-training; official model README","sections":[{"title":"How it works","body":"scFoundation embeds gene identity and continuous expression together with source and target read-depth indicators. Its encoder processes only nonzero, unmasked genes. Those contextual embeddings are combined with zero and mask embeddings before a Performer decoder predicts expression across the full gene vocabulary. Pooled encoder outputs represent cells; decoder outputs provide gene-level context.","source_ids":["evidence-official-bfbc8babf3630ddb3ed7","evidence-official-7972ce2bdd6a5e40d8ad","evidence-official-9863509ad10b824dc96b"],"source_locator":"June 15 2023 scFoundation manuscript: Results pre-training framework; Methods Data collection, Embedding, Encoder, Decoder and Read-depth-aware pre-training; official model README"},{"title":"Versions and reproducibility","body":"scFoundation / xTrimoscFoundation-alpha; the repository exposes separately configured embedding, enhancement and downstream prediction workflows. Fixed input gene vocabulary of 19,264 genes; not a nucleotide token context.","source_ids":["evidence-official-bfbc8babf3630ddb3ed7","evidence-official-7972ce2bdd6a5e40d8ad","evidence-official-9863509ad10b824dc96b"],"source_locator":"June 15 2023 scFoundation manuscript: Results pre-training framework; Methods Data collection, Embedding, Encoder, Decoder and Read-depth-aware pre-training; official model README"}],"facts":[{"label":"Model type","value":"Transcriptomic representation model","status":"source_checked","source_ids":["evidence-official-bfbc8babf3630ddb3ed7","evidence-official-7972ce2bdd6a5e40d8ad","evidence-official-9863509ad10b824dc96b"],"source_locator":"June 15 2023 scFoundation manuscript: Results pre-training framework; Methods Data collection, Embedding, Encoder, Decoder and Read-depth-aware pre-training; official model README"},{"label":"Architecture","value":"An asymmetric transformer encoder-decoder: learned continuous-expression embeddings enter a transformer encoder for nonzero, unmasked genes, then a Performer decoder reconstructs the full gene set.","status":"source_checked","source_ids":["evidence-official-bfbc8babf3630ddb3ed7","evidence-official-7972ce2bdd6a5e40d8ad","evidence-official-9863509ad10b824dc96b"],"source_locator":"June 15 2023 scFoundation manuscript: Results pre-training framework; Methods Data collection, Embedding, Encoder, Decoder and Read-depth-aware pre-training; official model README"},{"label":"Inputs","value":"Single-cell or bulk expression aligned to the documented 19,264-gene vocabulary.","status":"source_checked","source_ids":["evidence-official-bfbc8babf3630ddb3ed7","evidence-official-7972ce2bdd6a5e40d8ad","evidence-official-9863509ad10b824dc96b"],"source_locator":"June 15 2023 scFoundation manuscript: Results pre-training framework; Methods Data collection, Embedding, Encoder, Decoder and Read-depth-aware pre-training; official model README"},{"label":"Outputs","value":"Cell embeddings, contextual gene embeddings and outputs of separately configured downstream models.","status":"source_checked","source_ids":["evidence-official-bfbc8babf3630ddb3ed7","evidence-official-7972ce2bdd6a5e40d8ad","evidence-official-9863509ad10b824dc96b"],"source_locator":"June 15 2023 scFoundation manuscript: Results pre-training framework; Methods Data collection, Embedding, Encoder, Decoder and Read-depth-aware pre-training; official model README"},{"label":"Parameters","value":"100 million.","status":"source_checked","source_ids":["evidence-official-bfbc8babf3630ddb3ed7","evidence-official-7972ce2bdd6a5e40d8ad","evidence-official-9863509ad10b824dc96b"],"source_locator":"June 15 2023 scFoundation manuscript: Results pre-training framework; Methods Data collection, Embedding, Encoder, Decoder and Read-depth-aware pre-training; official model README"},{"label":"Known versions","value":"scFoundation / xTrimoscFoundation-alpha; the repository exposes separately configured embedding, enhancement and downstream prediction workflows.","status":"source_checked","source_ids":["evidence-official-bfbc8babf3630ddb3ed7","evidence-official-7972ce2bdd6a5e40d8ad","evidence-official-9863509ad10b824dc96b"],"source_locator":"June 15 2023 scFoundation manuscript: Results pre-training framework; Methods Data collection, Embedding, Encoder, Decoder and Read-depth-aware pre-training; official model README"},{"label":"Training data","value":"The June 2023 manuscript describes more than 50M human cells collected from GEO, Single Cell Portal, HCA and EMBL-EBI, aligned to 19,264 genes after quality control.","status":"source_checked","source_ids":["evidence-official-bfbc8babf3630ddb3ed7","evidence-official-7972ce2bdd6a5e40d8ad","evidence-official-9863509ad10b824dc96b"],"source_locator":"June 15 2023 scFoundation manuscript: Results pre-training framework; Methods Data collection, Embedding, Encoder, Decoder and Read-depth-aware pre-training; official model README"},{"label":"Training cutoff","value":"The June 2023 manuscript lists GEO, Single Cell Portal, HCA and EMBL-EBI as collection sources; its data-collection section does not give a shared last-included-study date.","status":"unreported","source_ids":["evidence-official-bfbc8babf3630ddb3ed7","evidence-official-7972ce2bdd6a5e40d8ad","evidence-official-9863509ad10b824dc96b"],"source_locator":"June 15 2023 scFoundation manuscript: Results pre-training framework; Methods Data collection, Embedding, Encoder, Decoder and Read-depth-aware pre-training; official model README"},{"label":"Context limits","value":"Fixed input gene vocabulary of 19,264 genes; not a nucleotide token context.","status":"source_checked","source_ids":["evidence-official-bfbc8babf3630ddb3ed7","evidence-official-7972ce2bdd6a5e40d8ad","evidence-official-9863509ad10b824dc96b"],"source_locator":"June 15 2023 scFoundation manuscript: Results pre-training framework; Methods Data collection, Embedding, Encoder, Decoder and Read-depth-aware pre-training; official model README"},{"label":"Weights licence","value":"Separate Model License; the Apache source-code notice explicitly excludes model-weight rights.","status":"source_checked","source_ids":["evidence-official-bfbc8babf3630ddb3ed7","evidence-official-7972ce2bdd6a5e40d8ad","evidence-official-9863509ad10b824dc96b"],"source_locator":"June 15 2023 scFoundation manuscript: Results pre-training framework; Methods Data collection, Embedding, Encoder, Decoder and Read-depth-aware pre-training; official model README"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/biomap-research/scFoundation","status":"source_checked","source_ids":["evidence-official-bfbc8babf3630ddb3ed7","evidence-official-7972ce2bdd6a5e40d8ad","evidence-official-9863509ad10b824dc96b"],"source_locator":"June 15 2023 scFoundation manuscript: Results pre-training framework; Methods Data collection, Embedding, Encoder, Decoder and Read-depth-aware pre-training; official model README"},{"label":"Code licence","value":"Apache-2.0","status":"source_checked","source_ids":["evidence-official-6189052c702a948da02d"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"The released workflow includes embedding extraction, read-depth enhancement and integration with downstream predictors.","source_ids":["evidence-official-bfbc8babf3630ddb3ed7","evidence-official-7972ce2bdd6a5e40d8ad","evidence-official-9863509ad10b824dc96b"],"source_locator":"June 15 2023 scFoundation manuscript: Results pre-training framework; Methods Data collection, Embedding, Encoder, Decoder and Read-depth-aware pre-training; official model README"}],"limitations":[{"text":"Gene identifiers must be aligned to the supplied gene index. Source-code licensing does not grant the separate model-weight rights.","source_ids":["evidence-official-bfbc8babf3630ddb3ed7","evidence-official-7972ce2bdd6a5e40d8ad","evidence-official-9863509ad10b824dc96b"],"source_locator":"June 15 2023 scFoundation manuscript: Results pre-training framework; Methods Data collection, Embedding, Encoder, Decoder and Read-depth-aware pre-training; official model README"}],"diagram":{"title":"scFoundation workflow","steps":["Expression and read-depth indicators","Gene and continuous-value embeddings","Sparse-input transformer encoder","Full-gene Performer decoder","Cell/gene embeddings or expression"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-bfbc8babf3630ddb3ed7","evidence-official-7972ce2bdd6a5e40d8ad","evidence-official-9863509ad10b824dc96b"],"source_locator":"June 15 2023 scFoundation manuscript: Results pre-training framework; Methods Data collection, Embedding, Encoder, Decoder and Read-depth-aware pre-training; official model README"},"coverage":"limited","gaps":["Training cutoff: The June 2023 manuscript lists GEO, Single Cell Portal, HCA and EMBL-EBI as collection sources; its data-collection section does not give a shared last-included-study date."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}}} {"id":"catalog-model-scgpt","kind":"model","name":"scGPT","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["cells-tissues"],"method_types":["foundation model"]},"source_ids":["catalog-source-scgpt"],"links":[{"relation":"variant_of","target_id":"discovery-model-scgpt"}],"attributes":{"entity_level":"family","version":"whole-human","reported_name":"scGPT","access":"Public code and downloadable checkpoints; use the unfine-tuned whole-human model for a new task.","method_type":"foundation model","historical_missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"scGPT learns representations of single-cell molecular measurements and supports task-specific adaptation.","summary_source_ids":["evidence-official-0e4dac85aa0e0ff1d4b2","evidence-official-d8e2d4f2e752370b6565","evidence-official-efa6ec9d886a21534a20"],"summary_source_locator":"May 2023 scGPT preprint Sections 4.1–4.3 and 4.8; current scgpt/model/model.py and README pretrained model table; README.md: Pretrained scGPT Model Zoo, whole-human row","sections":[{"title":"How it works","body":"scGPT combines each gene identity with its expression-value encoding before transformer attention. The implementation supports several expression encoders and cell-pooling choices. Task heads then predict expression or cell labels; optional masking and batch objectives depend on the training configuration.","source_ids":["evidence-official-0e4dac85aa0e0ff1d4b2","evidence-official-d8e2d4f2e752370b6565","evidence-official-efa6ec9d886a21534a20"],"source_locator":"May 2023 scGPT preprint Sections 4.1–4.3 and 4.8; current scgpt/model/model.py and README pretrained model table; README.md: Pretrained scGPT Model Zoo, whole-human row"},{"title":"Versions and reproducibility","body":"The May 2023 preprint reports an early 10M-cell model. The current whole-human checkpoint table reports 33M normal human cells; these sources describe different releases. The applicable input limits require configuration-specific checking.","source_ids":["evidence-official-0e4dac85aa0e0ff1d4b2","evidence-official-d8e2d4f2e752370b6565","evidence-official-efa6ec9d886a21534a20"],"source_locator":"May 2023 scGPT preprint Sections 4.1–4.3 and 4.8; current scgpt/model/model.py and README pretrained model table; README.md: Pretrained scGPT Model Zoo, whole-human row"}],"facts":[{"label":"Model type","value":"Generative single-cell transformer","status":"source_checked","source_ids":["evidence-official-0e4dac85aa0e0ff1d4b2","evidence-official-d8e2d4f2e752370b6565","evidence-official-efa6ec9d886a21534a20"],"source_locator":"May 2023 scGPT preprint Sections 4.1–4.3 and 4.8; current scgpt/model/model.py and README pretrained model table; README.md: Pretrained scGPT Model Zoo, whole-human row"},{"label":"Architecture","value":"Transformer backbone combining learned gene-token embeddings with expression-value encodings and optional batch encodings. Separate expression, cell-classification and optional masked-value or batch-discriminator heads support configured tasks.","status":"source_checked","source_ids":["evidence-official-0e4dac85aa0e0ff1d4b2","evidence-official-d8e2d4f2e752370b6565","evidence-official-efa6ec9d886a21534a20"],"source_locator":"May 2023 scGPT preprint Sections 4.1–4.3 and 4.8; current scgpt/model/model.py and README pretrained model table; README.md: Pretrained scGPT Model Zoo, whole-human row"},{"label":"Inputs","value":"Gene-expression measurements with the checkpoint-matched gene vocabulary.","status":"source_checked","source_ids":["evidence-official-0e4dac85aa0e0ff1d4b2","evidence-official-d8e2d4f2e752370b6565","evidence-official-efa6ec9d886a21534a20"],"source_locator":"May 2023 scGPT preprint Sections 4.1–4.3 and 4.8; current scgpt/model/model.py and README pretrained model table; README.md: Pretrained scGPT Model Zoo, whole-human row"},{"label":"Outputs","value":"Cell/gene representations and task-specific predictions after the relevant workflow.","status":"source_checked","source_ids":["evidence-official-0e4dac85aa0e0ff1d4b2","evidence-official-d8e2d4f2e752370b6565","evidence-official-efa6ec9d886a21534a20"],"source_locator":"May 2023 scGPT preprint Sections 4.1–4.3 and 4.8; current scgpt/model/model.py and README pretrained model table; README.md: Pretrained scGPT Model Zoo, whole-human row"},{"label":"Parameters","value":"The May 2023 model has 12 transformer blocks, width 512 and eight heads. The inspected current model-zoo table does not state the exact parameter total of its separate 33M-cell checkpoint.","status":"unreported","source_ids":["evidence-official-0e4dac85aa0e0ff1d4b2","evidence-official-d8e2d4f2e752370b6565","evidence-official-efa6ec9d886a21534a20"],"source_locator":"May 2023 scGPT preprint Sections 4.1–4.3 and 4.8; current scgpt/model/model.py and README pretrained model table; README.md: Pretrained scGPT Model Zoo, whole-human row"},{"label":"Known versions","value":"The May 2023 preprint reports an early 10M-cell model. The current whole-human checkpoint table reports 33M normal human cells; these sources describe different releases.","status":"source_checked","source_ids":["evidence-official-0e4dac85aa0e0ff1d4b2","evidence-official-d8e2d4f2e752370b6565","evidence-official-efa6ec9d886a21534a20"],"source_locator":"May 2023 scGPT preprint Sections 4.1–4.3 and 4.8; current scgpt/model/model.py and README pretrained model table; README.md: Pretrained scGPT Model Zoo, whole-human row"},{"label":"Training data","value":"The current whole-human model-zoo checkpoint uses 33M normal human cells, alongside separately released organ-specific and pan-cancer models. The earlier May 2023 preprint describes 10M training cells; its corpus is not the current checkpoint corpus.","status":"source_checked","source_ids":["evidence-official-0e4dac85aa0e0ff1d4b2","evidence-official-d8e2d4f2e752370b6565","evidence-official-efa6ec9d886a21534a20"],"source_locator":"May 2023 scGPT preprint Sections 4.1–4.3 and 4.8; current scgpt/model/model.py and README pretrained model table; README.md: Pretrained scGPT Model Zoo, whole-human row"},{"label":"Training cutoff","value":"The reviewed early manuscript and current whole-human model-zoo entry describe different corpora; neither supplies a shared latest-study date for the current checkpoint.","status":"unreported","source_ids":["evidence-official-0e4dac85aa0e0ff1d4b2","evidence-official-d8e2d4f2e752370b6565","evidence-official-efa6ec9d886a21534a20"],"source_locator":"May 2023 scGPT preprint Sections 4.1–4.3 and 4.8; current scgpt/model/model.py and README pretrained model table; README.md: Pretrained scGPT Model Zoo, whole-human row"},{"label":"Context limits","value":"The implementation accepts variable gene sets matched to its vocabulary. The reviewed model-zoo entry does not specify one validated maximum gene sequence for the current whole-human checkpoint.","status":"unreported","source_ids":["evidence-official-0e4dac85aa0e0ff1d4b2","evidence-official-d8e2d4f2e752370b6565","evidence-official-efa6ec9d886a21534a20"],"source_locator":"May 2023 scGPT preprint Sections 4.1–4.3 and 4.8; current scgpt/model/model.py and README pretrained model table; README.md: Pretrained scGPT Model Zoo, whole-human row"},{"label":"Weights licence","value":"Separate checkpoint-distribution terms are not stated in the inspected release documentation and licence material. The source-code licence alone is not recorded as an explicit weight grant.","status":"unreported","source_ids":["evidence-official-0e4dac85aa0e0ff1d4b2","evidence-official-d8e2d4f2e752370b6565","evidence-official-efa6ec9d886a21534a20","evidence-official-43484d4de29aacd65ed7"],"source_locator":"May 2023 scGPT preprint Sections 4.1–4.3 and 4.8; current scgpt/model/model.py and README pretrained model table; README.md: Pretrained scGPT Model Zoo, whole-human row; LICENSE: licence text"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/bowang-lab/scGPT","status":"source_checked","source_ids":["evidence-official-0e4dac85aa0e0ff1d4b2","evidence-official-d8e2d4f2e752370b6565","evidence-official-efa6ec9d886a21534a20"],"source_locator":"May 2023 scGPT preprint Sections 4.1–4.3 and 4.8; current scgpt/model/model.py and README pretrained model table; README.md: Pretrained scGPT Model Zoo, whole-human row"},{"label":"Code licence","value":"MIT","status":"source_checked","source_ids":["evidence-official-43484d4de29aacd65ed7"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"Releases include whole-human and organ-specific checkpoints, with tutorials for reference mapping and other downstream tasks.","source_ids":["evidence-official-0e4dac85aa0e0ff1d4b2","evidence-official-d8e2d4f2e752370b6565","evidence-official-efa6ec9d886a21534a20"],"source_locator":"May 2023 scGPT preprint Sections 4.1–4.3 and 4.8; current scgpt/model/model.py and README pretrained model table; README.md: Pretrained scGPT Model Zoo, whole-human row"}],"limitations":[{"text":"Checkpoint choice and vocabulary must match the biological context. A whole-human pretrained encoder and a fine-tuned annotation or perturbation model are distinct evaluated configurations.","source_ids":["evidence-official-0e4dac85aa0e0ff1d4b2","evidence-official-d8e2d4f2e752370b6565","evidence-official-efa6ec9d886a21534a20"],"source_locator":"May 2023 scGPT preprint Sections 4.1–4.3 and 4.8; current scgpt/model/model.py and README pretrained model table; README.md: Pretrained scGPT Model Zoo, whole-human row"}],"diagram":{"title":"scGPT workflow","steps":["Gene expression and vocabulary","scGPT encoder","Cell and gene representations","Task-specific adaptation"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-0e4dac85aa0e0ff1d4b2","evidence-official-d8e2d4f2e752370b6565","evidence-official-efa6ec9d886a21534a20"],"source_locator":"May 2023 scGPT preprint Sections 4.1–4.3 and 4.8; current scgpt/model/model.py and README pretrained model table; README.md: Pretrained scGPT Model Zoo, whole-human row"},"coverage":"limited","gaps":["Parameters: The May 2023 model has 12 transformer blocks, width 512 and eight heads. The inspected current model-zoo table does not state the exact parameter total of its separate 33M-cell checkpoint.","Training cutoff: The reviewed early manuscript and current whole-human model-zoo entry describe different corpora; neither supplies a shared latest-study date for the current checkpoint.","Context limits: The implementation accepts variable gene sets matched to its vocabulary. The reviewed model-zoo entry does not specify one validated maximum gene sequence for the current whole-human checkpoint.","Weights licence: Separate checkpoint-distribution terms are not stated in the inspected release documentation and licence material. The source-code licence alone is not recorded as an explicit weight grant."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}}} {"id":"catalog-model-scvi","kind":"model","name":"scVI","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["cells-tissues"],"method_types":["baseline"]},"source_ids":["catalog-source-scvi"],"links":[],"attributes":{"entity_level":"family","version":"scvi-tools","reported_name":"scVI","access":"Public software; train a task-specific model on the permitted split.","method_type":"baseline","historical_missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"scVI models single-cell RNA counts with a probabilistic latent-variable model that accounts for observed covariates.","summary_source_ids":["evidence-official-550f6719c2f22f4d57b3","evidence-official-0cbab80c2d1769287a1f"],"summary_source_locator":"docs/user_guide/models/scvi.md: Preliminaries, Generative process, Inference and Limitations","sections":[{"title":"How it works","body":"scVI models single-cell RNA counts with a probabilistic latent-variable model that accounts for observed covariates. Variational autoencoder with a count likelihood and neural encoder/decoder; likelihood and batch/dispersion settings are configurable. The documented inputs are cell-by-gene count matrix, optionally with batch, donor or other covariates. The output consists of low-dimensional cell representations, normalized expression and probabilistic downstream quantities.","source_ids":["evidence-official-550f6719c2f22f4d57b3","evidence-official-0cbab80c2d1769287a1f"],"source_locator":"docs/user_guide/models/scvi.md: Preliminaries, Generative process, Inference and Limitations"},{"title":"Versions and reproducibility","body":"scVI model within scvi-tools; package version, likelihood, covariates and checkpoint are evaluation-specific. Gene-feature matrix rather than a fixed sequence-token window.","source_ids":["evidence-official-550f6719c2f22f4d57b3","evidence-official-0cbab80c2d1769287a1f"],"source_locator":"docs/user_guide/models/scvi.md: Preliminaries, Generative process, Inference and Limitations"}],"facts":[{"label":"Model type","value":"Variational autoencoder for count data","status":"source_checked","source_ids":["evidence-official-550f6719c2f22f4d57b3","evidence-official-0cbab80c2d1769287a1f"],"source_locator":"docs/user_guide/models/scvi.md: Preliminaries, Generative process, Inference and Limitations"},{"label":"Architecture","value":"Variational autoencoder with a count likelihood and neural encoder/decoder; likelihood and batch/dispersion settings are configurable.","status":"source_checked","source_ids":["evidence-official-550f6719c2f22f4d57b3","evidence-official-0cbab80c2d1769287a1f"],"source_locator":"docs/user_guide/models/scvi.md: Preliminaries, Generative process, Inference and Limitations"},{"label":"Inputs","value":"Cell-by-gene count matrix, optionally with batch, donor or other covariates.","status":"source_checked","source_ids":["evidence-official-550f6719c2f22f4d57b3","evidence-official-0cbab80c2d1769287a1f"],"source_locator":"docs/user_guide/models/scvi.md: Preliminaries, Generative process, Inference and Limitations"},{"label":"Outputs","value":"Low-dimensional cell representations, normalized expression and probabilistic downstream quantities.","status":"source_checked","source_ids":["evidence-official-550f6719c2f22f4d57b3","evidence-official-0cbab80c2d1769287a1f"],"source_locator":"docs/user_guide/models/scvi.md: Preliminaries, Generative process, Inference and Limitations"},{"label":"Parameters","value":"Configuration-dependent, including gene count and encoder/decoder dimensions.","status":"source_checked","source_ids":["evidence-official-550f6719c2f22f4d57b3","evidence-official-0cbab80c2d1769287a1f"],"source_locator":"docs/user_guide/models/scvi.md: Preliminaries, Generative process, Inference and Limitations"},{"label":"Known versions","value":"scVI model within scvi-tools; package version, likelihood, covariates and checkpoint are evaluation-specific.","status":"source_checked","source_ids":["evidence-official-550f6719c2f22f4d57b3","evidence-official-0cbab80c2d1769287a1f"],"source_locator":"docs/user_guide/models/scvi.md: Preliminaries, Generative process, Inference and Limitations"},{"label":"Training data","value":"Fitted to the user-selected count matrix or a specified pretrained reference; scVI is not one universal checkpoint.","status":"source_checked","source_ids":["evidence-official-550f6719c2f22f4d57b3","evidence-official-0cbab80c2d1769287a1f"],"source_locator":"docs/user_guide/models/scvi.md: Preliminaries, Generative process, Inference and Limitations"},{"label":"Training cutoff","value":"Inapplicable as one universal pretraining date: scVI is fitted to the supplied dataset, whose collection date and train/test split belong to the evaluation.","status":"inapplicable","source_ids":["evidence-official-550f6719c2f22f4d57b3","evidence-official-0cbab80c2d1769287a1f"],"source_locator":"docs/user_guide/models/scvi.md: Preliminaries, Generative process, Inference and Limitations"},{"label":"Context limits","value":"Gene-feature matrix rather than a fixed sequence-token window.","status":"source_checked","source_ids":["evidence-official-550f6719c2f22f4d57b3","evidence-official-0cbab80c2d1769287a1f"],"source_locator":"docs/user_guide/models/scvi.md: Preliminaries, Generative process, Inference and Limitations"},{"label":"Weights licence","value":"No universal weights release applies to a model fitted on each dataset; any reused checkpoint requires its own licence.","status":"source_checked","source_ids":["evidence-official-550f6719c2f22f4d57b3","evidence-official-0cbab80c2d1769287a1f"],"source_locator":"docs/user_guide/models/scvi.md: Preliminaries, Generative process, Inference and Limitations"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/scverse/scvi-tools","status":"source_checked","source_ids":["evidence-official-550f6719c2f22f4d57b3","evidence-official-0cbab80c2d1769287a1f"],"source_locator":"docs/user_guide/models/scvi.md: Preliminaries, Generative process, Inference and Limitations"},{"label":"Code licence","value":"BSD-3-Clause","status":"source_checked","source_ids":["evidence-official-6851724e3bcb7e9d2781"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"Models count observations directly and supports batch-conditioned expression estimates and reference-to-query transfer.","source_ids":["evidence-official-550f6719c2f22f4d57b3","evidence-official-0cbab80c2d1769287a1f"],"source_locator":"docs/user_guide/models/scvi.md: Preliminaries, Generative process, Inference and Limitations"}],"limitations":[{"text":"The documentation notes that the latent space is less interpretable than a linear method and efficient inference generally benefits from a GPU. Covariates and likelihood must be reported.","source_ids":["evidence-official-550f6719c2f22f4d57b3","evidence-official-0cbab80c2d1769287a1f"],"source_locator":"docs/user_guide/models/scvi.md: Preliminaries, Generative process, Inference and Limitations"}],"diagram":{"title":"scVI workflow","steps":["RNA counts and covariates","Variational encoder","Latent cell state","Count decoder and estimates"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-550f6719c2f22f4d57b3","evidence-official-0cbab80c2d1769287a1f"],"source_locator":"docs/user_guide/models/scvi.md: Preliminaries, Generative process, Inference and Limitations"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}}} {"id":"catalog-model-spliceai","kind":"model","name":"SpliceAI","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["dna-genomes"],"method_types":["specialist"]},"source_ids":["catalog-source-spliceai"],"links":[{"relation":"variant_of","target_id":"discovery-model-spliceai"}],"attributes":{"entity_level":"family","version":"1.3.1","reported_name":"SpliceAI","access":"Public archived code under PolyForm Strict; model weights are CC BY-NC 4.0 for non-commercial use.","method_type":"specialist","historical_missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"SpliceAI annotates sequence variants with predicted splice acceptor and donor changes.","summary_source_ids":["evidence-official-864a2c5ea6e61aa310f9","evidence-official-ded281404bd1a0f3fdb7"],"summary_source_locator":"Jaganathan et al., Cell 2019, STAR Methods: SpliceAI architecture and Model training and testing (pp. e2–e3); README.md: usage and License","sections":[{"title":"How it works","body":"SpliceAI reads one-hot-encoded DNA through dilated convolutional residual blocks. Skip connections combine features at different depths, and a softmax layer assigns acceptor, donor or neither probabilities to the central positions. The 10kb version requires 5kb of sequence on each side of a scored position; variant scoring compares the reference and alternate predictions.","source_ids":["evidence-official-864a2c5ea6e61aa310f9","evidence-official-ded281404bd1a0f3fdb7"],"source_locator":"Jaganathan et al., Cell 2019, STAR Methods: SpliceAI architecture and Model training and testing (pp. e2–e3); README.md: usage and License"},{"title":"Versions and reproducibility","body":"The paper studies 80nt, 400nt, 2kb and 10kb receptive spans. Variant scoring averages five independently trained models; these are not five different assay results. SpliceAI-10k uses 5,000 flanking bases on each side. An input of length l + 10,000 produces predictions for l central positions; receptive span is distinct from maximum input length.","source_ids":["evidence-official-864a2c5ea6e61aa310f9","evidence-official-ded281404bd1a0f3fdb7"],"source_locator":"Jaganathan et al., Cell 2019, STAR Methods: SpliceAI architecture and Model training and testing (pp. e2–e3); README.md: usage and License"}],"facts":[{"label":"Model type","value":"Dilated convolutional splicing predictor","status":"source_checked","source_ids":["evidence-official-864a2c5ea6e61aa310f9","evidence-official-ded281404bd1a0f3fdb7"],"source_locator":"Jaganathan et al., Cell 2019, STAR Methods: SpliceAI architecture and Model training and testing (pp. e2–e3); README.md: usage and License"},{"label":"Architecture","value":"Residual one-dimensional convolutional network with dilated kernels and skip connections; a softmax head predicts acceptor, donor and neither at each central position.","status":"source_checked","source_ids":["evidence-official-864a2c5ea6e61aa310f9","evidence-official-ded281404bd1a0f3fdb7"],"source_locator":"Jaganathan et al., Cell 2019, STAR Methods: SpliceAI architecture and Model training and testing (pp. e2–e3); README.md: usage and License"},{"label":"Inputs","value":"VCF variants, reference FASTA and matching gene annotation, or custom one-hot-encoded sequence.","status":"source_checked","source_ids":["evidence-official-864a2c5ea6e61aa310f9","evidence-official-ded281404bd1a0f3fdb7"],"source_locator":"Jaganathan et al., Cell 2019, STAR Methods: SpliceAI architecture and Model training and testing (pp. e2–e3); README.md: usage and License"},{"label":"Outputs","value":"Acceptor/donor gain/loss scores and positions in VCF INFO annotations.","status":"source_checked","source_ids":["evidence-official-864a2c5ea6e61aa310f9","evidence-official-ded281404bd1a0f3fdb7"],"source_locator":"Jaganathan et al., Cell 2019, STAR Methods: SpliceAI architecture and Model training and testing (pp. e2–e3); README.md: usage and License"},{"label":"Parameters","value":"The original STAR Methods specifies residual blocks, dilation and receptive spans, but does not state a complete parameter count for each released five-model scoring ensemble.","status":"unreported","source_ids":["evidence-official-864a2c5ea6e61aa310f9","evidence-official-ded281404bd1a0f3fdb7"],"source_locator":"Jaganathan et al., Cell 2019, STAR Methods: SpliceAI architecture and Model training and testing (pp. e2–e3); README.md: usage and License"},{"label":"Known versions","value":"The paper studies 80nt, 400nt, 2kb and 10kb receptive spans. Variant scoring averages five independently trained models; these are not five different assay results.","status":"source_checked","source_ids":["evidence-official-864a2c5ea6e61aa310f9","evidence-official-ded281404bd1a0f3fdb7"],"source_locator":"Jaganathan et al., Cell 2019, STAR Methods: SpliceAI architecture and Model training and testing (pp. e2–e3); README.md: usage and License"},{"label":"Training data","value":"Human GRCh37 sequence and GENCODE V24lift37 principal protein-coding transcripts, split by chromosome with non-paralogous held-out test genes. The paper distinguishes GENCODE-only training from GTEx-junction-augmented models used for variant analyses.","status":"source_checked","source_ids":["evidence-official-864a2c5ea6e61aa310f9","evidence-official-ded281404bd1a0f3fdb7"],"source_locator":"Jaganathan et al., Cell 2019, STAR Methods: SpliceAI architecture and Model training and testing (pp. e2–e3); README.md: usage and License"},{"label":"Training cutoff","value":"GENCODE V24lift37 on GRCh37 defines the documented transcript annotations. GTEx-augmented training is separately described; the paper does not give one common latest-data date for both variants.","status":"source_checked","source_ids":["evidence-official-864a2c5ea6e61aa310f9","evidence-official-ded281404bd1a0f3fdb7"],"source_locator":"Jaganathan et al., Cell 2019, STAR Methods: SpliceAI architecture and Model training and testing (pp. e2–e3); README.md: usage and License"},{"label":"Context limits","value":"SpliceAI-10k uses 5,000 flanking bases on each side. An input of length l + 10,000 produces predictions for l central positions; receptive span is distinct from maximum input length.","status":"source_checked","source_ids":["evidence-official-864a2c5ea6e61aa310f9","evidence-official-ded281404bd1a0f3fdb7"],"source_locator":"Jaganathan et al., Cell 2019, STAR Methods: SpliceAI architecture and Model training and testing (pp. e2–e3); README.md: usage and License"},{"label":"Weights licence","value":"CC-BY-NC-4.0 for trained models; commercial use requires a separate licence. Code is PolyForm Strict 1.0.0.","status":"source_checked","source_ids":["evidence-official-864a2c5ea6e61aa310f9","evidence-official-ded281404bd1a0f3fdb7"],"source_locator":"Jaganathan et al., Cell 2019, STAR Methods: SpliceAI architecture and Model training and testing (pp. e2–e3); README.md: usage and License"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/Illumina/SpliceAI","status":"source_checked","source_ids":["evidence-official-864a2c5ea6e61aa310f9","evidence-official-ded281404bd1a0f3fdb7"],"source_locator":"Jaganathan et al., Cell 2019, STAR Methods: SpliceAI architecture and Model training and testing (pp. e2–e3); README.md: usage and License"},{"label":"Code licence","value":"PolyForm Strict 1.0.0 for code; trained weights have separate terms.","status":"source_checked","source_ids":["evidence-official-c465f6d4fcc04f9afe4b"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"Provides direct sequence inference and an annotation workflow with explicit genome and distance settings.","source_ids":["evidence-official-864a2c5ea6e61aa310f9","evidence-official-ded281404bd1a0f3fdb7"],"source_locator":"Jaganathan et al., Cell 2019, STAR Methods: SpliceAI architecture and Model training and testing (pp. e2–e3); README.md: usage and License"}],"limitations":[{"text":"The command-line pipeline skips unsupported variants and variants outside its gene annotations. Code, model weights and downloadable precomputed scores have distinct licensing provisions.","source_ids":["evidence-official-864a2c5ea6e61aa310f9","evidence-official-ded281404bd1a0f3fdb7"],"source_locator":"Jaganathan et al., Cell 2019, STAR Methods: SpliceAI architecture and Model training and testing (pp. e2–e3); README.md: usage and License"}],"diagram":{"title":"SpliceAI workflow","steps":["Variant plus sequence context","Reference and alternate predictions","Splice-site differences","Gain/loss annotations"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-864a2c5ea6e61aa310f9","evidence-official-ded281404bd1a0f3fdb7"],"source_locator":"Jaganathan et al., Cell 2019, STAR Methods: SpliceAI architecture and Model training and testing (pp. e2–e3); README.md: usage and License"},"coverage":"limited","gaps":["Parameters: The original STAR Methods specifies residual blocks, dilation and receptive spans, but does not state a complete parameter count for each released five-model scoring ensemble."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}}} {"id":"catalog-model-vina","kind":"model","name":"AutoDock Vina","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["molecular-interactions"],"method_types":["baseline"]},"source_ids":["catalog-source-vina"],"links":[],"attributes":{"entity_level":"family","version":"1.2.7","reported_name":"AutoDock Vina","access":"Public docking software; receptor and ligand preparation required.","method_type":"baseline","historical_missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"AutoDock Vina searches for ligand conformations and poses in a molecular docking problem.","summary_source_ids":["evidence-official-23e72f7e93a0a633ab2b"],"summary_source_locator":"README.md: introduction, feature list, license and Citations","sections":[{"title":"How it works","body":"AutoDock Vina searches for ligand conformations and poses in a molecular docking problem. Scoring functions coupled to gradient-based conformational optimization and search. The documented inputs are prepared receptor and ligand structures with the configured search space and scoring function. The output consists of candidate docking poses and corresponding docking scores.","source_ids":["evidence-official-23e72f7e93a0a633ab2b"],"source_locator":"README.md: introduction, feature list, license and Citations"},{"title":"Versions and reproducibility","body":"AutoDock Vina; README cites the 1.2.0 feature expansion separately from the original 2010 method. Molecular geometry and search-box constraints rather than a sequence context window.","source_ids":["evidence-official-23e72f7e93a0a633ab2b"],"source_locator":"README.md: introduction, feature list, license and Citations"}],"facts":[{"label":"Model type","value":"Classical molecular docking software","status":"source_checked","source_ids":["evidence-official-23e72f7e93a0a633ab2b"],"source_locator":"README.md: introduction, feature list, license and Citations"},{"label":"Architecture","value":"Scoring functions coupled to gradient-based conformational optimization and search.","status":"source_checked","source_ids":["evidence-official-23e72f7e93a0a633ab2b"],"source_locator":"README.md: introduction, feature list, license and Citations"},{"label":"Inputs","value":"Prepared receptor and ligand structures with the configured search space and scoring function.","status":"source_checked","source_ids":["evidence-official-23e72f7e93a0a633ab2b"],"source_locator":"README.md: introduction, feature list, license and Citations"},{"label":"Outputs","value":"Candidate docking poses and corresponding docking scores.","status":"source_checked","source_ids":["evidence-official-23e72f7e93a0a633ab2b"],"source_locator":"README.md: introduction, feature list, license and Citations"},{"label":"Parameters","value":"Inapplicable as a neural parameter count; scoring/search parameters are separately configured.","status":"inapplicable","source_ids":["evidence-official-23e72f7e93a0a633ab2b"],"source_locator":"README.md: introduction, feature list, license and Citations"},{"label":"Known versions","value":"AutoDock Vina; README cites the 1.2.0 feature expansion separately from the original 2010 method.","status":"source_checked","source_ids":["evidence-official-23e72f7e93a0a633ab2b"],"source_locator":"README.md: introduction, feature list, license and Citations"},{"label":"Training data","value":"Not a pretrained neural model. The selected scoring function and parametrization define the procedural baseline.","status":"source_checked","source_ids":["evidence-official-23e72f7e93a0a633ab2b"],"source_locator":"README.md: introduction, feature list, license and Citations"},{"label":"Training cutoff","value":"Inapplicable to neural pretraining; reference-database and input-data dates must be recorded for each run.","status":"inapplicable","source_ids":["evidence-official-23e72f7e93a0a633ab2b"],"source_locator":"README.md: introduction, feature list, license and Citations"},{"label":"Context limits","value":"Molecular geometry and search-box constraints rather than a sequence context window.","status":"source_checked","source_ids":["evidence-official-23e72f7e93a0a633ab2b"],"source_locator":"README.md: introduction, feature list, license and Citations"},{"label":"Weights licence","value":"Inapplicable: no neural model-weight checkpoint.","status":"inapplicable","source_ids":["evidence-official-23e72f7e93a0a633ab2b"],"source_locator":"README.md: introduction, feature list, license and Citations"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/ccsb-scripps/AutoDock-Vina","status":"source_checked","source_ids":["evidence-official-23e72f7e93a0a633ab2b"],"source_locator":"README.md: introduction, feature list, license and Citations"},{"label":"Code licence","value":"Apache-2.0","status":"source_checked","source_ids":["evidence-official-640665f30e62eed9319b"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"Supports Vina and AutoDock4 scoring, multiple-ligand/batch workflows, macrocycles and Python bindings.","source_ids":["evidence-official-23e72f7e93a0a633ab2b"],"source_locator":"README.md: introduction, feature list, license and Citations"}],"limitations":[{"text":"The chosen scoring function, molecular preparation and search configuration form part of the method. A docking score is a computational quantity rather than an experimental affinity measurement.","source_ids":["evidence-official-23e72f7e93a0a633ab2b"],"source_locator":"README.md: introduction, feature list, license and Citations"}],"diagram":{"title":"AutoDock Vina workflow","steps":["Prepared receptor and ligand","Conformational search","Docking score","Ranked candidate poses"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-23e72f7e93a0a633ab2b"],"source_locator":"README.md: introduction, feature list, license and Citations"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}}} {"id":"catalog-source-alphafold-3-server","kind":"source","name":"AlphaFold 3 Server official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://alphafoldserver.com/output-terms","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-alphagenome","kind":"source","name":"AlphaGenome official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/google-deepmind/alphagenome_research","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-boltz-2","kind":"source","name":"Boltz-2 official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/jwohlwend/boltz","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-chai-1","kind":"source","name":"Chai-1 official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/chaidiscovery/chai-lab","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-diffdock-l","kind":"source","name":"DiffDock-L official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/gcorso/DiffDock","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-dnabert-2","kind":"source","name":"DNABERT-2 official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://huggingface.co/zhihan1996/DNABERT-2-117M","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-esm-2","kind":"source","name":"ESM-2 official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/facebookresearch/esm","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-esmfold","kind":"source","name":"ESMFold official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/facebookresearch/esm","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-evo-2","kind":"source","name":"Evo 2 official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/ArcInstitute/evo2","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-gears","kind":"source","name":"GEARS official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/snap-stanford/GEARS","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-geneformer","kind":"source","name":"Geneformer official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://huggingface.co/ctheodoris/Geneformer","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-kraken2","kind":"source","name":"Kraken2 official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/DerrickWood/kraken2","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-metagene-1","kind":"source","name":"METAGENE-1 official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://huggingface.co/metagene-ai/METAGENE-1","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-metaphlan","kind":"source","name":"MetaPhlAn official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/biobakery/MetaPhlAn","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-mimic","kind":"source","name":"MIMIC official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://huggingface.co/polymathic-ai/MIMIC","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-mrna-fm","kind":"source","name":"mRNA-FM official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/ml4bio/RNA-FM","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-nt-v2","kind":"source","name":"Nucleotide Transformer v2 official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://huggingface.co/InstaDeepAI/nucleotide-transformer-v2-50m-multi-species","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-pangolin","kind":"source","name":"Pangolin official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/tkzeng/Pangolin","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-prokbert","kind":"source","name":"ProkBERT official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/nbrg-ppcu/prokbert","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-proteinmpnn","kind":"source","name":"ProteinMPNN official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/dauparas/ProteinMPNN","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-rhofold","kind":"source","name":"RhoFold+ official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/ml4bio/RhoFold","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-rna-fm","kind":"source","name":"RNA-FM official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/ml4bio/RNA-FM","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-scfoundation","kind":"source","name":"scFoundation official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/biomap-research/scFoundation","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-scgpt","kind":"source","name":"scGPT official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/bowang-lab/scGPT","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-scvi","kind":"source","name":"scVI official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/scverse/scvi-tools","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-spliceai","kind":"source","name":"SpliceAI official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/Illumina/SpliceAI","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-vina","kind":"source","name":"AutoDock Vina official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/ccsb-scripps/AutoDock-Vina","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-task-cell-batch-integration","kind":"benchmark","name":"Batch integration","description":"","status":"discovered","facets":{"areas":["cells-tissues"]},"source_ids":["catalog-source-scgpt","catalog-source-scvi"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"Batch integration","scope_note":"Test whether cell identity is retained across donors and batches.","historical_missing_metadata":{"protocol_version":"not_yet_extracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"Batch integration asks whether measurements from different experiments can be combined while preserving biological differences.","summary_source_ids":["src-discovery-theislab-scib"],"summary_source_locator":"README: Metrics, Biological Conservation, Batch Correction and Integration Tools","sections":[{"title":"Choosing an evaluation","body":"Assess two objectives separately: removal of technical batch effects and retention of cell identities and biological variation. scIB supplies metrics for both. A model that mixes every cell together can score well on mixing while destroying useful biology, so one objective cannot substitute for the other. Use a concrete scIB or Open Problems protocol for datasets, preprocessing and scoring; those resources are not interchangeable.","source_ids":["src-discovery-theislab-scib"],"source_locator":"README: Metrics, Biological Conservation, Batch Correction and Integration Tools"},{"title":"Task scope","body":"This is a task guide, not a single versioned benchmark protocol. The connected resources provide examples or concrete procedures. A candidate method or proposed control is not evidence that an evaluation has been completed.","source_ids":["src-discovery-theislab-scib"],"source_locator":"README: Metrics, Biological Conservation, Batch Correction and Integration Tools"}],"facts":[{"label":"Entity type","value":"Task guide; concrete protocol identities remain separate.","status":"source_checked","source_ids":["src-discovery-theislab-scib"],"source_locator":"README: Metrics, Biological Conservation, Batch Correction and Integration Tools"},{"label":"Datasets","value":"No single dataset is fixed by this guide. Select a linked protocol and its versioned data release.","status":"inapplicable","source_ids":["src-discovery-theislab-scib"],"source_locator":"README: Metrics, Biological Conservation, Batch Correction and Integration Tools"},{"label":"Organisms","value":"No shared organism population is defined at this guide level. Record it for each selected dataset.","status":"inapplicable","source_ids":["src-discovery-theislab-scib"],"source_locator":"README: Metrics, Biological Conservation, Batch Correction and Integration Tools"},{"label":"Assays","value":"No single measurement assay is fixed by this guide; the endpoint and assay belong to the selected protocol.","status":"inapplicable","source_ids":["src-discovery-theislab-scib"],"source_locator":"README: Metrics, Biological Conservation, Batch Correction and Integration Tools"},{"label":"Splits","value":"No executable split is attached to this task identity. Use the selected protocol’s split manifest.","status":"inapplicable","source_ids":["src-discovery-theislab-scib"],"source_locator":"README: Metrics, Biological Conservation, Batch Correction and Integration Tools"},{"label":"Allowed inputs","value":"Single-cell measurements with batch labels and biological annotations; the chosen method may return a corrected matrix, embedding or graph.","status":"source_checked","source_ids":["src-discovery-theislab-scib"],"source_locator":"README: Metrics, Biological Conservation, Batch Correction and Integration Tools"},{"label":"Adaptation","value":"No common fitting regime is imposed here. Keep pretrained, frozen, probed, fine-tuned and conventional methods distinct where applicable.","status":"inapplicable","source_ids":["src-discovery-theislab-scib"],"source_locator":"README: Metrics, Biological Conservation, Batch Correction and Integration Tools"},{"label":"Metrics","value":"scIB includes cell-type silhouette, ARI/NMI and trajectory conservation for biology; batch silhouette, iLISI, kBET and graph connectivity for batch effects.","status":"source_checked","source_ids":["src-discovery-theislab-scib"],"source_locator":"README: Metrics, Biological Conservation, Batch Correction and Integration Tools"},{"label":"Baselines","value":"Documented integration comparators include Harmony, MNN, scVI and Seurat. Versions and preprocessing are part of the configuration.","status":"source_checked","source_ids":["src-discovery-theislab-scib"],"source_locator":"README: Metrics, Biological Conservation, Batch Correction and Integration Tools"}],"strengths":[{"text":"A two-part assessment can expose overcorrection that a batch-mixing score alone misses.","source_ids":["src-discovery-theislab-scib"],"source_locator":"README: Metrics, Biological Conservation, Batch Correction and Integration Tools"}],"limitations":[{"text":"Metric choice depends on available labels and output representation. This guide supplies neither a shared cohort nor a universal combined score.","source_ids":["src-discovery-theislab-scib"],"source_locator":"README: Metrics, Biological Conservation, Batch Correction and Integration Tools"}],"diagram":{"title":"Conceptual evaluation workflow","steps":["Select annotated datasets and protocol","Apply the integration method","Measure retained biological variation","Measure residual batch effects"],"caption":"Conceptual task guide. Dataset preparation, parameters and scoring must come from a separately identified protocol.","source_ids":["src-discovery-theislab-scib"],"source_locator":"README: Metrics, Biological Conservation, Batch Correction and Integration Tools"},"coverage":"reviewed","gaps":["A concrete evaluation still needs a protocol version, dataset release, eligible/scored counts and documented uncertainty. This guide does not manufacture those run-specific values."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed named primary documentation and existing protocol records. Editorial task scope is distinguished from source-reported procedures. No model execution or new numerical result; a reviewed guide is not a complete runnable protocol."}}}} {"id":"catalog-task-cell-perturbation","kind":"benchmark","name":"Perturbation response","description":"","status":"discovered","facets":{"areas":["cells-tissues"]},"source_ids":["catalog-source-geneformer","catalog-source-scfoundation","catalog-source-gears"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"Perturbation response","scope_note":"Predict expression changes after unseen perturbations.","historical_missing_metadata":{"protocol_version":"not_yet_extracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"Perturbation-response prediction asks whether a model can predict molecular measurements under a changed cellular condition.","summary_source_ids":["src-discovery-altoslabs-perturbench"],"summary_source_locator":"README: Downloading and Preparing Datasets, Data Splitting, Evaluator Class and Automated Evaluation","sections":[{"title":"Choosing an evaluation","body":"A concrete protocol defines which conditions are observed during fitting and which are held out. Compare predicted and measured responses with matched covariates and a declared aggregation. PerturBench separates average expression, changes from controls and distributional comparisons; these answer different questions. Report the dataset-specific split rather than assuming that a random cell split tests unseen perturbations.","source_ids":["src-discovery-altoslabs-perturbench"],"source_locator":"README: Downloading and Preparing Datasets, Data Splitting, Evaluator Class and Automated Evaluation"},{"title":"Task scope","body":"This is a task guide, not a single versioned benchmark protocol. The connected resources provide examples or concrete procedures. A candidate method or proposed control is not evidence that an evaluation has been completed.","source_ids":["src-discovery-altoslabs-perturbench"],"source_locator":"README: Downloading and Preparing Datasets, Data Splitting, Evaluator Class and Automated Evaluation"}],"facts":[{"label":"Entity type","value":"Task guide; concrete protocol identities remain separate.","status":"source_checked","source_ids":["src-discovery-altoslabs-perturbench"],"source_locator":"README: Downloading and Preparing Datasets, Data Splitting, Evaluator Class and Automated Evaluation"},{"label":"Datasets","value":"No single dataset is fixed by this guide. Select a linked protocol and its versioned data release.","status":"inapplicable","source_ids":["src-discovery-altoslabs-perturbench"],"source_locator":"README: Downloading and Preparing Datasets, Data Splitting, Evaluator Class and Automated Evaluation"},{"label":"Organisms","value":"No shared organism population is defined at this guide level. Record it for each selected dataset.","status":"inapplicable","source_ids":["src-discovery-altoslabs-perturbench"],"source_locator":"README: Downloading and Preparing Datasets, Data Splitting, Evaluator Class and Automated Evaluation"},{"label":"Assays","value":"No single measurement assay is fixed by this guide; the endpoint and assay belong to the selected protocol.","status":"inapplicable","source_ids":["src-discovery-altoslabs-perturbench"],"source_locator":"README: Downloading and Preparing Datasets, Data Splitting, Evaluator Class and Automated Evaluation"},{"label":"Splits","value":"No executable split is attached to this task identity. Use the selected protocol’s split manifest.","status":"inapplicable","source_ids":["src-discovery-altoslabs-perturbench"],"source_locator":"README: Downloading and Preparing Datasets, Data Splitting, Evaluator Class and Automated Evaluation"},{"label":"Allowed inputs","value":"Control-cell measurements, requested condition and permitted covariates; predictions and observed responses aligned to the same features.","status":"source_checked","source_ids":["src-discovery-altoslabs-perturbench"],"source_locator":"README: Downloading and Preparing Datasets, Data Splitting, Evaluator Class and Automated Evaluation"},{"label":"Adaptation","value":"No common fitting regime is imposed here. Keep pretrained, frozen, probed, fine-tuned and conventional methods distinct where applicable.","status":"inapplicable","source_ids":["src-discovery-altoslabs-perturbench"],"source_locator":"README: Downloading and Preparing Datasets, Data Splitting, Evaluator Class and Automated Evaluation"},{"label":"Metrics","value":"PerturBench supports expression-error, correlation, change-from-control and distributional metrics; aggregation and ranking settings are explicit configuration choices.","status":"source_checked","source_ids":["src-discovery-altoslabs-perturbench"],"source_locator":"README: Downloading and Preparing Datasets, Data Splitting, Evaluator Class and Automated Evaluation"},{"label":"Baselines","value":"Choose the baselines supplied by the selected protocol and record their access to controls and training conditions. No baseline result is asserted by this guide.","status":"source_checked","source_ids":["src-discovery-altoslabs-perturbench"],"source_locator":"README: Downloading and Preparing Datasets, Data Splitting, Evaluator Class and Automated Evaluation"}],"strengths":[{"text":"Matched controls and held-out conditions allow an evaluation to separate reconstruction from generalisation to a new condition.","source_ids":["src-discovery-altoslabs-perturbench"],"source_locator":"README: Downloading and Preparing Datasets, Data Splitting, Evaluator Class and Automated Evaluation"}],"limitations":[{"text":"A good score on average expression need not establish correct changes or cell-to-cell variation. Dose, time and cell context must match.","source_ids":["src-discovery-altoslabs-perturbench"],"source_locator":"README: Downloading and Preparing Datasets, Data Splitting, Evaluator Class and Automated Evaluation"}],"diagram":{"title":"Conceptual evaluation workflow","steps":["Select data and held-out conditions","Specify allowed controls and covariates","Predict molecular responses","Score declared aggregates and distributions"],"caption":"Conceptual task guide. Dataset preparation, parameters and scoring must come from a separately identified protocol.","source_ids":["src-discovery-altoslabs-perturbench"],"source_locator":"README: Downloading and Preparing Datasets, Data Splitting, Evaluator Class and Automated Evaluation"},"coverage":"reviewed","gaps":["A concrete evaluation still needs a protocol version, dataset release, eligible/scored counts and documented uncertainty. This guide does not manufacture those run-specific values."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed named primary documentation and existing protocol records. Editorial task scope is distinguished from source-reported procedures. No model execution or new numerical result; a reviewed guide is not a complete runnable protocol."}}}} {"id":"catalog-task-cell-reference-mapping","kind":"benchmark","name":"Donor-held-out reference mapping","description":"","status":"discovered","facets":{"areas":["cells-tissues"]},"source_ids":["catalog-source-geneformer","catalog-source-scgpt","catalog-source-scfoundation","catalog-source-scvi"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"Donor-held-out reference mapping","scope_note":"Map unseen donors to a labelled cell-type reference.","historical_missing_metadata":{"protocol_version":"not_yet_extracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"Donor-held-out reference mapping asks whether a reference learned from some donors remains useful for cells from a new donor.","summary_source_ids":["evidence-official-0cbab80c2d1769287a1f","src-discovery-theislab-scib"],"summary_source_locator":"scVI model guide: generative model and latent representation; scIB README: biological-conservation metrics. Donor holdout is this guide’s evaluation requirement, not an asserted scVI default.","sections":[{"title":"Choosing an evaluation","body":"Keep donor identity visible throughout the evaluation. The reference and any fitted annotation rules belong to the training data; a query donor supplies the held-out assessment. Cell representations and transferred labels are different outputs and require different scoring. This is a proposed task definition, not a claim that every reference-mapping paper uses donor holdout.","source_ids":["evidence-official-0cbab80c2d1769287a1f","src-discovery-theislab-scib"],"source_locator":"scVI model guide: generative model and latent representation; scIB README: biological-conservation metrics. Donor holdout is this guide’s evaluation requirement, not an asserted scVI default."},{"title":"Task scope","body":"This is a task guide, not a single versioned benchmark protocol. The connected resources provide examples or concrete procedures. A candidate method or proposed control is not evidence that an evaluation has been completed.","source_ids":["evidence-official-0cbab80c2d1769287a1f","src-discovery-theislab-scib"],"source_locator":"scVI model guide: generative model and latent representation; scIB README: biological-conservation metrics. Donor holdout is this guide’s evaluation requirement, not an asserted scVI default."}],"facts":[{"label":"Entity type","value":"Task guide; concrete protocol identities remain separate.","status":"source_checked","source_ids":["evidence-official-0cbab80c2d1769287a1f","src-discovery-theislab-scib"],"source_locator":"scVI model guide: generative model and latent representation; scIB README: biological-conservation metrics. Donor holdout is this guide’s evaluation requirement, not an asserted scVI default."},{"label":"Datasets","value":"No single dataset is fixed by this guide. Select a linked protocol and its versioned data release.","status":"inapplicable","source_ids":["evidence-official-0cbab80c2d1769287a1f","src-discovery-theislab-scib"],"source_locator":"scVI model guide: generative model and latent representation; scIB README: biological-conservation metrics. Donor holdout is this guide’s evaluation requirement, not an asserted scVI default."},{"label":"Organisms","value":"No shared organism population is defined at this guide level. Record it for each selected dataset.","status":"inapplicable","source_ids":["evidence-official-0cbab80c2d1769287a1f","src-discovery-theislab-scib"],"source_locator":"scVI model guide: generative model and latent representation; scIB README: biological-conservation metrics. Donor holdout is this guide’s evaluation requirement, not an asserted scVI default."},{"label":"Assays","value":"No single measurement assay is fixed by this guide; the endpoint and assay belong to the selected protocol.","status":"inapplicable","source_ids":["evidence-official-0cbab80c2d1769287a1f","src-discovery-theislab-scib"],"source_locator":"scVI model guide: generative model and latent representation; scIB README: biological-conservation metrics. Donor holdout is this guide’s evaluation requirement, not an asserted scVI default."},{"label":"Splits","value":"This proposed guide requires donor separation; no particular donor assignment or cohort is fixed here.","status":"inapplicable","source_ids":["evidence-official-0cbab80c2d1769287a1f","src-discovery-theislab-scib"],"source_locator":"scVI model guide: generative model and latent representation; scIB README: biological-conservation metrics. Donor holdout is this guide’s evaluation requirement, not an asserted scVI default."},{"label":"Allowed inputs","value":"Reference and query molecular measurements, compatible feature identities, donor metadata and labels reserved for assessment.","status":"source_checked","source_ids":["evidence-official-0cbab80c2d1769287a1f","src-discovery-theislab-scib"],"source_locator":"scVI model guide: generative model and latent representation; scIB README: biological-conservation metrics. Donor holdout is this guide’s evaluation requirement, not an asserted scVI default."},{"label":"Adaptation","value":"No common fitting regime is imposed here. Keep pretrained, frozen, probed, fine-tuned and conventional methods distinct where applicable.","status":"inapplicable","source_ids":["evidence-official-0cbab80c2d1769287a1f","src-discovery-theislab-scib"],"source_locator":"scVI model guide: generative model and latent representation; scIB README: biological-conservation metrics. Donor holdout is this guide’s evaluation requirement, not an asserted scVI default."},{"label":"Metrics","value":"Choose label-transfer metrics for annotations and biological-conservation metrics for representations. The exact metric, averaging and unknown-cell handling belong to the concrete protocol.","status":"source_checked","source_ids":["evidence-official-0cbab80c2d1769287a1f","src-discovery-theislab-scib"],"source_locator":"scVI model guide: generative model and latent representation; scIB README: biological-conservation metrics. Donor holdout is this guide’s evaluation requirement, not an asserted scVI default."},{"label":"Baselines","value":"Specify a reference-only comparator with the same available labels and features. No measured baseline is assigned here.","status":"source_checked","source_ids":["evidence-official-0cbab80c2d1769287a1f","src-discovery-theislab-scib"],"source_locator":"scVI model guide: generative model and latent representation; scIB README: biological-conservation metrics. Donor holdout is this guide’s evaluation requirement, not an asserted scVI default."}],"strengths":[{"text":"Separating donors makes the intended generalisation question explicit and avoids confusing held-out cells with a new biological replicate.","source_ids":["evidence-official-0cbab80c2d1769287a1f","src-discovery-theislab-scib"],"source_locator":"scVI model guide: generative model and latent representation; scIB README: biological-conservation metrics. Donor holdout is this guide’s evaluation requirement, not an asserted scVI default."}],"limitations":[{"text":"A donor split alone does not control every difference in assay, tissue or cell composition. This guide does not define a particular cohort.","source_ids":["evidence-official-0cbab80c2d1769287a1f","src-discovery-theislab-scib"],"source_locator":"scVI model guide: generative model and latent representation; scIB README: biological-conservation metrics. Donor holdout is this guide’s evaluation requirement, not an asserted scVI default."}],"diagram":{"title":"Conceptual evaluation workflow","steps":["Define reference and query donors","Fit reference using permitted data","Map held-out query cells","Score labels and biological structure"],"caption":"Conceptual task guide. Dataset preparation, parameters and scoring must come from a separately identified protocol.","source_ids":["evidence-official-0cbab80c2d1769287a1f","src-discovery-theislab-scib"],"source_locator":"scVI model guide: generative model and latent representation; scIB README: biological-conservation metrics. Donor holdout is this guide’s evaluation requirement, not an asserted scVI default."},"coverage":"reviewed","gaps":["A concrete evaluation still needs a protocol version, dataset release, eligible/scored counts and documented uncertainty. This guide does not manufacture those run-specific values."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed named primary documentation and existing protocol records. Editorial task scope is distinguished from source-reported procedures. No model execution or new numerical result; a reviewed guide is not a complete runnable protocol."}}}} {"id":"catalog-task-community-profiling","kind":"benchmark","name":"Community profiling","description":"","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":["catalog-source-metaphlan"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"Community profiling","scope_note":"Estimate taxon abundances in metagenomic samples.","historical_missing_metadata":{"protocol_version":"not_yet_extracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"Community profiling estimates which microorganisms are present in a sample and their relative abundances.","summary_source_ids":["src-discovery-cami-challenge-opal"],"summary_source_locator":"README: Overview, Computed metrics and Inputs","sections":[{"title":"Choosing an evaluation","body":"Compare a predicted taxonomic profile with a specified reference profile at declared taxonomic ranks. OPAL separates detection errors from abundance errors and diversity summaries. The reference taxonomy, filtering and abundance normalisation must accompany a result; changing these can change what is scored.","source_ids":["src-discovery-cami-challenge-opal"],"source_locator":"README: Overview, Computed metrics and Inputs"},{"title":"Task scope","body":"This is a task guide, not a single versioned benchmark protocol. The connected resources provide examples or concrete procedures. A candidate method or proposed control is not evidence that an evaluation has been completed.","source_ids":["src-discovery-cami-challenge-opal"],"source_locator":"README: Overview, Computed metrics and Inputs"}],"facts":[{"label":"Entity type","value":"Task guide; concrete protocol identities remain separate.","status":"source_checked","source_ids":["src-discovery-cami-challenge-opal"],"source_locator":"README: Overview, Computed metrics and Inputs"},{"label":"Datasets","value":"No single dataset is fixed by this guide. Select a linked protocol and its versioned data release.","status":"inapplicable","source_ids":["src-discovery-cami-challenge-opal"],"source_locator":"README: Overview, Computed metrics and Inputs"},{"label":"Organisms","value":"No shared organism population is defined at this guide level. Record it for each selected dataset.","status":"inapplicable","source_ids":["src-discovery-cami-challenge-opal"],"source_locator":"README: Overview, Computed metrics and Inputs"},{"label":"Assays","value":"No single measurement assay is fixed by this guide; the endpoint and assay belong to the selected protocol.","status":"inapplicable","source_ids":["src-discovery-cami-challenge-opal"],"source_locator":"README: Overview, Computed metrics and Inputs"},{"label":"Splits","value":"No executable split is attached to this task identity. Use the selected protocol’s split manifest.","status":"inapplicable","source_ids":["src-discovery-cami-challenge-opal"],"source_locator":"README: Overview, Computed metrics and Inputs"},{"label":"Allowed inputs","value":"Predicted and reference taxonomic profiles with consistent taxon identifiers and abundance definitions.","status":"source_checked","source_ids":["src-discovery-cami-challenge-opal"],"source_locator":"README: Overview, Computed metrics and Inputs"},{"label":"Adaptation","value":"No common fitting regime is imposed here. Keep pretrained, frozen, probed, fine-tuned and conventional methods distinct where applicable.","status":"inapplicable","source_ids":["src-discovery-cami-challenge-opal"],"source_locator":"README: Overview, Computed metrics and Inputs"},{"label":"Metrics","value":"OPAL reports precision, recall, F1, abundance errors such as L1 and Bray–Curtis, and diversity measures. These are distinct endpoints.","status":"source_checked","source_ids":["src-discovery-cami-challenge-opal"],"source_locator":"README: Overview, Computed metrics and Inputs"},{"label":"Baselines","value":"Compare profilers against the same reference profile and taxonomic ranks; OPAL is an evaluator, not a predictive baseline.","status":"source_checked","source_ids":["src-discovery-cami-challenge-opal"],"source_locator":"README: Overview, Computed metrics and Inputs"}],"strengths":[{"text":"Separate presence and abundance metrics reveal different failure modes instead of hiding them behind a single accuracy number.","source_ids":["src-discovery-cami-challenge-opal"],"source_locator":"README: Overview, Computed metrics and Inputs"}],"limitations":[{"text":"A profile-level result is not a read-binning result. Reference coverage and taxonomic rank affect interpretation.","source_ids":["src-discovery-cami-challenge-opal"],"source_locator":"README: Overview, Computed metrics and Inputs"}],"diagram":{"title":"Conceptual evaluation workflow","steps":["Fix reference profile and taxonomy","Generate candidate abundance profiles","Align taxa and ranks","Score detection and abundance separately"],"caption":"Conceptual task guide. Dataset preparation, parameters and scoring must come from a separately identified protocol.","source_ids":["src-discovery-cami-challenge-opal"],"source_locator":"README: Overview, Computed metrics and Inputs"},"coverage":"reviewed","gaps":["A concrete evaluation still needs a protocol version, dataset release, eligible/scored counts and documented uncertainty. This guide does not manufacture those run-specific values."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed named primary documentation and existing protocol records. Editorial task scope is distinguished from source-reported procedures. No model execution or new numerical result; a reviewed guide is not a complete runnable protocol."}}}} {"id":"catalog-task-complex-structure","kind":"benchmark","name":"Biomolecular complex structure","description":"","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":["catalog-source-chai-1","catalog-source-boltz-2","catalog-source-alphafold-3-server"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"Biomolecular complex structure","scope_note":"Predict joint structure for interacting proteins and other molecules.","historical_missing_metadata":{"protocol_version":"not_yet_extracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"Complex-structure evaluation tests predicted molecular arrangements against experimentally determined structures.","summary_source_ids":["evidence-alphafold-paper"],"summary_source_locator":"Main text: Model architecture and Model limitations; Methods: Metrics and Recent PDB evaluation set","sections":[{"title":"Choosing an evaluation","body":"Declare the molecular partners and the input information available to each method. Assess the relevant chains and interfaces as well as the whole complex, using an explicit chain-assignment rule. Templates, alignments and sampling budgets belong to the evaluated configuration. A confidence estimate is not an experimental accuracy measurement.","source_ids":["evidence-alphafold-paper"],"source_locator":"Main text: Model architecture and Model limitations; Methods: Metrics and Recent PDB evaluation set"},{"title":"Task scope","body":"This is a task guide, not a single versioned benchmark protocol. The connected resources provide examples or concrete procedures. A candidate method or proposed control is not evidence that an evaluation has been completed.","source_ids":["evidence-alphafold-paper"],"source_locator":"Main text: Model architecture and Model limitations; Methods: Metrics and Recent PDB evaluation set"}],"facts":[{"label":"Entity type","value":"Task guide; concrete protocol identities remain separate.","status":"source_checked","source_ids":["evidence-alphafold-paper"],"source_locator":"Main text: Model architecture and Model limitations; Methods: Metrics and Recent PDB evaluation set"},{"label":"Datasets","value":"No single dataset is fixed by this guide. Select a linked protocol and its versioned data release.","status":"inapplicable","source_ids":["evidence-alphafold-paper"],"source_locator":"Main text: Model architecture and Model limitations; Methods: Metrics and Recent PDB evaluation set"},{"label":"Organisms","value":"No shared organism population is defined at this guide level. Record it for each selected dataset.","status":"inapplicable","source_ids":["evidence-alphafold-paper"],"source_locator":"Main text: Model architecture and Model limitations; Methods: Metrics and Recent PDB evaluation set"},{"label":"Assays","value":"No single measurement assay is fixed by this guide; the endpoint and assay belong to the selected protocol.","status":"inapplicable","source_ids":["evidence-alphafold-paper"],"source_locator":"Main text: Model architecture and Model limitations; Methods: Metrics and Recent PDB evaluation set"},{"label":"Splits","value":"No executable split is attached to this task identity. Use the selected protocol’s split manifest.","status":"inapplicable","source_ids":["evidence-alphafold-paper"],"source_locator":"Main text: Model architecture and Model limitations; Methods: Metrics and Recent PDB evaluation set"},{"label":"Allowed inputs","value":"Molecular sequences and chemical identities, with templates or alignments only where the protocol permits them; experimental reference structures for scoring.","status":"source_checked","source_ids":["evidence-alphafold-paper"],"source_locator":"Main text: Model architecture and Model limitations; Methods: Metrics and Recent PDB evaluation set"},{"label":"Adaptation","value":"No common fitting regime is imposed here. Keep pretrained, frozen, probed, fine-tuned and conventional methods distinct where applicable.","status":"inapplicable","source_ids":["evidence-alphafold-paper"],"source_locator":"Main text: Model architecture and Model limitations; Methods: Metrics and Recent PDB evaluation set"},{"label":"Metrics","value":"Protocol-specific structure and interface measures, such as LDDT, DockQ or interface LDDT. State the assessed entities and aggregation.","status":"source_checked","source_ids":["evidence-alphafold-paper"],"source_locator":"Main text: Model architecture and Model limitations; Methods: Metrics and Recent PDB evaluation set"},{"label":"Baselines","value":"Compare structure predictors with matched partner definitions and input information. Different sampling budgets require separate reporting.","status":"source_checked","source_ids":["evidence-alphafold-paper"],"source_locator":"Main text: Model architecture and Model limitations; Methods: Metrics and Recent PDB evaluation set"}],"strengths":[{"text":"Interface-specific scoring can reveal errors hidden by an otherwise accurate large component.","source_ids":["evidence-alphafold-paper"],"source_locator":"Main text: Model architecture and Model limitations; Methods: Metrics and Recent PDB evaluation set"}],"limitations":[{"text":"Agreement with one reference structure does not establish dynamics, binding affinity or every possible conformational state.","source_ids":["evidence-alphafold-paper"],"source_locator":"Main text: Model architecture and Model limitations; Methods: Metrics and Recent PDB evaluation set"}],"diagram":{"title":"Conceptual evaluation workflow","steps":["Fix partners, references and allowed inputs","Predict complete complexes","Match equivalent chains and atoms","Score structures and interfaces"],"caption":"Conceptual task guide. Dataset preparation, parameters and scoring must come from a separately identified protocol.","source_ids":["evidence-alphafold-paper"],"source_locator":"Main text: Model architecture and Model limitations; Methods: Metrics and Recent PDB evaluation set"},"coverage":"reviewed","gaps":["A concrete evaluation still needs a protocol version, dataset release, eligible/scored counts and documented uncertainty. This guide does not manufacture those run-specific values."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed named primary documentation and existing protocol records. Editorial task scope is distinguished from source-reported procedures. No model execution or new numerical result; a reviewed guide is not a complete runnable protocol."}}}} {"id":"catalog-task-enhancer-effects","kind":"benchmark","name":"Enhancer / MPRA effects","description":"","status":"discovered","facets":{"areas":["dna-genomes"]},"source_ids":["catalog-source-dnabert-2","catalog-source-nt-v2","catalog-source-evo-2","catalog-source-alphagenome"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"Enhancer / MPRA effects","scope_note":"Predict measured activity changes from regulatory sequence variants.","historical_missing_metadata":{"protocol_version":"not_yet_extracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"Enhancer-effect evaluation asks whether sequence-based predictions track measured changes in regulatory activity.","summary_source_ids":["src-discovery-kundajelab-dart-eval","dart-eval-regulatory-2024"],"summary_source_locator":"DART-Eval README Task 5; paper evaluation tasks and ab initio baselines","sections":[{"title":"Choosing an evaluation","body":"Separate detection of regulatory DNA from prediction of the effect of a sequence change. A reporter assay and endogenous chromatin-accessibility assay are different endpoints. DART-Eval provides concrete regulatory prediction tasks and comparisons across frozen, probed and fine-tuned models. Use its exact task and data release rather than treating every enhancer test as equivalent.","source_ids":["src-discovery-kundajelab-dart-eval","dart-eval-regulatory-2024"],"source_locator":"DART-Eval README Task 5; paper evaluation tasks and ab initio baselines"},{"title":"Task scope","body":"This is a task guide, not a single versioned benchmark protocol. The connected resources provide examples or concrete procedures. A candidate method or proposed control is not evidence that an evaluation has been completed.","source_ids":["src-discovery-kundajelab-dart-eval","dart-eval-regulatory-2024"],"source_locator":"DART-Eval README Task 5; paper evaluation tasks and ab initio baselines"}],"facts":[{"label":"Entity type","value":"Task guide; concrete protocol identities remain separate.","status":"source_checked","source_ids":["src-discovery-kundajelab-dart-eval","dart-eval-regulatory-2024"],"source_locator":"DART-Eval README Task 5; paper evaluation tasks and ab initio baselines"},{"label":"Datasets","value":"No single dataset is fixed by this guide. Select a linked protocol and its versioned data release.","status":"inapplicable","source_ids":["src-discovery-kundajelab-dart-eval","dart-eval-regulatory-2024"],"source_locator":"DART-Eval README Task 5; paper evaluation tasks and ab initio baselines"},{"label":"Organisms","value":"No shared organism population is defined at this guide level. Record it for each selected dataset.","status":"inapplicable","source_ids":["src-discovery-kundajelab-dart-eval","dart-eval-regulatory-2024"],"source_locator":"DART-Eval README Task 5; paper evaluation tasks and ab initio baselines"},{"label":"Assays","value":"No single measurement assay is fixed by this guide; the endpoint and assay belong to the selected protocol.","status":"inapplicable","source_ids":["src-discovery-kundajelab-dart-eval","dart-eval-regulatory-2024"],"source_locator":"DART-Eval README Task 5; paper evaluation tasks and ab initio baselines"},{"label":"Splits","value":"No executable split is attached to this task identity. Use the selected protocol’s split manifest.","status":"inapplicable","source_ids":["src-discovery-kundajelab-dart-eval","dart-eval-regulatory-2024"],"source_locator":"DART-Eval README Task 5; paper evaluation tasks and ab initio baselines"},{"label":"Allowed inputs","value":"Regulatory DNA or matched alleles and a specified experimental activity endpoint, with genome assembly and window placement recorded.","status":"source_checked","source_ids":["src-discovery-kundajelab-dart-eval","dart-eval-regulatory-2024"],"source_locator":"DART-Eval README Task 5; paper evaluation tasks and ab initio baselines"},{"label":"Adaptation","value":"No common fitting regime is imposed here. Keep pretrained, frozen, probed, fine-tuned and conventional methods distinct where applicable.","status":"inapplicable","source_ids":["src-discovery-kundajelab-dart-eval","dart-eval-regulatory-2024"],"source_locator":"DART-Eval README Task 5; paper evaluation tasks and ab initio baselines"},{"label":"Metrics","value":"Use the metric attached to the concrete task: regulatory-element classification and quantitative effect prediction require different scoring.","status":"source_checked","source_ids":["src-discovery-kundajelab-dart-eval","dart-eval-regulatory-2024"],"source_locator":"DART-Eval README Task 5; paper evaluation tasks and ab initio baselines"},{"label":"Baselines","value":"DART-Eval includes supervised models trained from scratch, including ChromBPNet-related baselines, alongside language-model evaluation modes.","status":"source_checked","source_ids":["src-discovery-kundajelab-dart-eval","dart-eval-regulatory-2024"],"source_locator":"DART-Eval README Task 5; paper evaluation tasks and ab initio baselines"}],"strengths":[{"text":"A functional endpoint and matched learned baselines test usefulness beyond sequence-likelihood differences alone.","source_ids":["src-discovery-kundajelab-dart-eval","dart-eval-regulatory-2024"],"source_locator":"DART-Eval README Task 5; paper evaluation tasks and ab initio baselines"}],"limitations":[{"text":"Results from one reporter design or cellular context do not automatically transfer to endogenous regulation elsewhere.","source_ids":["src-discovery-kundajelab-dart-eval","dart-eval-regulatory-2024"],"source_locator":"DART-Eval README Task 5; paper evaluation tasks and ab initio baselines"}],"diagram":{"title":"Conceptual evaluation workflow","steps":["Choose assay endpoint and split","Define DNA windows and alleles","Apply declared prediction configuration","Compare scores with measured effects"],"caption":"Conceptual task guide. Dataset preparation, parameters and scoring must come from a separately identified protocol.","source_ids":["src-discovery-kundajelab-dart-eval","dart-eval-regulatory-2024"],"source_locator":"DART-Eval README Task 5; paper evaluation tasks and ab initio baselines"},"coverage":"reviewed","gaps":["A concrete evaluation still needs a protocol version, dataset release, eligible/scored counts and documented uncertainty. This guide does not manufacture those run-specific values."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed named primary documentation and existing protocol records. Editorial task scope is distinguished from source-reported procedures. No model execution or new numerical result; a reviewed guide is not a complete runnable protocol."}}}} {"id":"catalog-task-heldout-clade","kind":"benchmark","name":"Held-out-clade classification","description":"","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":["catalog-source-prokbert","catalog-source-metagene-1","catalog-source-evo-2","catalog-source-kraken2"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"Held-out-clade classification","scope_note":"Hold clades out of downstream fitting and reference databases; evaluate known ancestor labels or unknown-taxon detection, and audit pretraining overlap separately.","historical_missing_metadata":{"protocol_version":"not_yet_extracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"Held-out-clade classification asks how well sequence-based classification works when a defined taxonomic group is absent from training.","summary_source_ids":["barcodebert-2026"],"summary_source_locator":"Methods: dataset partition and 1-NN probing; Results: unseen-species genus prediction","sections":[{"title":"Choosing an evaluation","body":"Specify both the taxonomic level that is withheld and the level that is predicted. BarcodeBERT’s unseen-species genus probe illustrates why these differ: species can be unseen while their genera remain represented. A benchmark must document which reference sequences and labels remain available before interpreting a result as taxonomic generalisation.","source_ids":["barcodebert-2026"],"source_locator":"Methods: dataset partition and 1-NN probing; Results: unseen-species genus prediction"},{"title":"Task scope","body":"This is a task guide, not a single versioned benchmark protocol. The connected resources provide examples or concrete procedures. A candidate method or proposed control is not evidence that an evaluation has been completed.","source_ids":["barcodebert-2026"],"source_locator":"Methods: dataset partition and 1-NN probing; Results: unseen-species genus prediction"}],"facts":[{"label":"Entity type","value":"Task guide; concrete protocol identities remain separate.","status":"source_checked","source_ids":["barcodebert-2026"],"source_locator":"Methods: dataset partition and 1-NN probing; Results: unseen-species genus prediction"},{"label":"Datasets","value":"No single dataset is fixed by this guide. Select a linked protocol and its versioned data release.","status":"inapplicable","source_ids":["barcodebert-2026"],"source_locator":"Methods: dataset partition and 1-NN probing; Results: unseen-species genus prediction"},{"label":"Organisms","value":"No shared organism population is defined at this guide level. Record it for each selected dataset.","status":"inapplicable","source_ids":["barcodebert-2026"],"source_locator":"Methods: dataset partition and 1-NN probing; Results: unseen-species genus prediction"},{"label":"Assays","value":"No single measurement assay is fixed by this guide; the endpoint and assay belong to the selected protocol.","status":"inapplicable","source_ids":["barcodebert-2026"],"source_locator":"Methods: dataset partition and 1-NN probing; Results: unseen-species genus prediction"},{"label":"Splits","value":"No executable split is attached to this task identity. Use the selected protocol’s split manifest.","status":"inapplicable","source_ids":["barcodebert-2026"],"source_locator":"Methods: dataset partition and 1-NN probing; Results: unseen-species genus prediction"},{"label":"Allowed inputs","value":"Labelled reference sequences and a taxonomically held-out query set; exact rank and reference-library membership are protocol-specific.","status":"source_checked","source_ids":["barcodebert-2026"],"source_locator":"Methods: dataset partition and 1-NN probing; Results: unseen-species genus prediction"},{"label":"Adaptation","value":"No common fitting regime is imposed here. Keep pretrained, frozen, probed, fine-tuned and conventional methods distinct where applicable.","status":"inapplicable","source_ids":["barcodebert-2026"],"source_locator":"Methods: dataset partition and 1-NN probing; Results: unseen-species genus prediction"},{"label":"Metrics","value":"Classification accuracy or other declared label metrics at the target rank; results at different taxonomic ranks are not interchangeable.","status":"source_checked","source_ids":["barcodebert-2026"],"source_locator":"Methods: dataset partition and 1-NN probing; Results: unseen-species genus prediction"},{"label":"Baselines","value":"The BarcodeBERT protocol includes nearest-neighbour evaluation of sequence representations. A proposed alternative comparator is not a completed result.","status":"source_checked","source_ids":["barcodebert-2026"],"source_locator":"Methods: dataset partition and 1-NN probing; Results: unseen-species genus prediction"}],"strengths":[{"text":"Explicit taxonomic holdout distinguishes close-reference matching from transfer to less familiar sequences.","source_ids":["barcodebert-2026"],"source_locator":"Methods: dataset partition and 1-NN probing; Results: unseen-species genus prediction"}],"limitations":[{"text":"Unseen-species performance within known genera is not evidence for classifying entirely novel genera.","source_ids":["barcodebert-2026"],"source_locator":"Methods: dataset partition and 1-NN probing; Results: unseen-species genus prediction"}],"diagram":{"title":"Conceptual evaluation workflow","steps":["Choose held-out and predicted ranks","Construct reference and query sets","Apply the declared classifier or probe","Score at the specified taxonomic rank"],"caption":"Conceptual task guide. Dataset preparation, parameters and scoring must come from a separately identified protocol.","source_ids":["barcodebert-2026"],"source_locator":"Methods: dataset partition and 1-NN probing; Results: unseen-species genus prediction"},"coverage":"reviewed","gaps":["A concrete evaluation still needs a protocol version, dataset release, eligible/scored counts and documented uncertainty. This guide does not manufacture those run-specific values."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed named primary documentation and existing protocol records. Editorial task scope is distinguished from source-reported procedures. No model execution or new numerical result; a reviewed guide is not a complete runnable protocol."}}}} {"id":"catalog-task-ligand-affinity","kind":"benchmark","name":"Small-molecule affinity","description":"","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":["catalog-source-boltz-2"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"Small-molecule affinity","scope_note":"Predict measured binding affinity; pose confidence is not an affinity value.","historical_missing_metadata":{"protocol_version":"not_yet_extracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"Affinity evaluation tests predictions of molecular binding measurements or binder labels, depending on the protocol.","summary_source_ids":["evidence-official-5a9e55abdf288d9dd8da","src-discovery-plinder-org-plinder"],"summary_source_locator":"Boltz docs/prediction.md: Properties (affinity), Affinity prediction outputs; PLINDER README: Known bugs","sections":[{"title":"Choosing an evaluation","body":"Define the endpoint before comparing numbers. Boltz-2 distinguishes a binder-versus-decoy probability from a quantitative affinity output, with different supervision. Preserve measurement units, transforms and assay conditions. Structural plausibility alone does not validate affinity; a structural dataset also needs checked affinity labels before it can support this task.","source_ids":["evidence-official-5a9e55abdf288d9dd8da","src-discovery-plinder-org-plinder"],"source_locator":"Boltz docs/prediction.md: Properties (affinity), Affinity prediction outputs; PLINDER README: Known bugs"},{"title":"Task scope","body":"This is a task guide, not a single versioned benchmark protocol. The connected resources provide examples or concrete procedures. A candidate method or proposed control is not evidence that an evaluation has been completed.","source_ids":["evidence-official-5a9e55abdf288d9dd8da","src-discovery-plinder-org-plinder"],"source_locator":"Boltz docs/prediction.md: Properties (affinity), Affinity prediction outputs; PLINDER README: Known bugs"}],"facts":[{"label":"Entity type","value":"Task guide; concrete protocol identities remain separate.","status":"source_checked","source_ids":["evidence-official-5a9e55abdf288d9dd8da","src-discovery-plinder-org-plinder"],"source_locator":"Boltz docs/prediction.md: Properties (affinity), Affinity prediction outputs; PLINDER README: Known bugs"},{"label":"Datasets","value":"No single dataset is fixed by this guide. Select a linked protocol and its versioned data release.","status":"inapplicable","source_ids":["evidence-official-5a9e55abdf288d9dd8da","src-discovery-plinder-org-plinder"],"source_locator":"Boltz docs/prediction.md: Properties (affinity), Affinity prediction outputs; PLINDER README: Known bugs"},{"label":"Organisms","value":"No shared organism population is defined at this guide level. Record it for each selected dataset.","status":"inapplicable","source_ids":["evidence-official-5a9e55abdf288d9dd8da","src-discovery-plinder-org-plinder"],"source_locator":"Boltz docs/prediction.md: Properties (affinity), Affinity prediction outputs; PLINDER README: Known bugs"},{"label":"Assays","value":"No single measurement assay is fixed by this guide; the endpoint and assay belong to the selected protocol.","status":"inapplicable","source_ids":["evidence-official-5a9e55abdf288d9dd8da","src-discovery-plinder-org-plinder"],"source_locator":"Boltz docs/prediction.md: Properties (affinity), Affinity prediction outputs; PLINDER README: Known bugs"},{"label":"Splits","value":"No executable split is attached to this task identity. Use the selected protocol’s split manifest.","status":"inapplicable","source_ids":["evidence-official-5a9e55abdf288d9dd8da","src-discovery-plinder-org-plinder"],"source_locator":"Boltz docs/prediction.md: Properties (affinity), Affinity prediction outputs; PLINDER README: Known bugs"},{"label":"Allowed inputs","value":"A protein–small-molecule pair plus any permitted structural information; matched experimental measurements or declared binary labels.","status":"source_checked","source_ids":["evidence-official-5a9e55abdf288d9dd8da","src-discovery-plinder-org-plinder"],"source_locator":"Boltz docs/prediction.md: Properties (affinity), Affinity prediction outputs; PLINDER README: Known bugs"},{"label":"Adaptation","value":"No common fitting regime is imposed here. Keep pretrained, frozen, probed, fine-tuned and conventional methods distinct where applicable.","status":"inapplicable","source_ids":["evidence-official-5a9e55abdf288d9dd8da","src-discovery-plinder-org-plinder"],"source_locator":"Boltz docs/prediction.md: Properties (affinity), Affinity prediction outputs; PLINDER README: Known bugs"},{"label":"Metrics","value":"Use endpoint-appropriate regression or classification metrics. Do not merge differently transformed affinity quantities into one table without documenting conversions.","status":"source_checked","source_ids":["evidence-official-5a9e55abdf288d9dd8da","src-discovery-plinder-org-plinder"],"source_locator":"Boltz docs/prediction.md: Properties (affinity), Affinity prediction outputs; PLINDER README: Known bugs"},{"label":"Baselines","value":"Select comparators evaluated on the same measured endpoint and split; a docking score is not automatically a calibrated affinity measurement.","status":"source_checked","source_ids":["evidence-official-5a9e55abdf288d9dd8da","src-discovery-plinder-org-plinder"],"source_locator":"Boltz docs/prediction.md: Properties (affinity), Affinity prediction outputs; PLINDER README: Known bugs"}],"strengths":[{"text":"Separating binding classification from quantitative affinity avoids conflating two different uses of a model output.","source_ids":["evidence-official-5a9e55abdf288d9dd8da","src-discovery-plinder-org-plinder"],"source_locator":"Boltz docs/prediction.md: Properties (affinity), Affinity prediction outputs; PLINDER README: Known bugs"}],"limitations":[{"text":"The inspected PLINDER revision flags its affinity query as disabled after a parsing bug. A dataset name alone does not verify an affinity label.","source_ids":["evidence-official-5a9e55abdf288d9dd8da","src-discovery-plinder-org-plinder"],"source_locator":"Boltz docs/prediction.md: Properties (affinity), Affinity prediction outputs; PLINDER README: Known bugs"}],"diagram":{"title":"Conceptual evaluation workflow","steps":["Fix assay endpoint, units and split","Specify model inputs","Predict the declared affinity quantity","Score matched measurements or labels"],"caption":"Conceptual task guide. Dataset preparation, parameters and scoring must come from a separately identified protocol.","source_ids":["evidence-official-5a9e55abdf288d9dd8da","src-discovery-plinder-org-plinder"],"source_locator":"Boltz docs/prediction.md: Properties (affinity), Affinity prediction outputs; PLINDER README: Known bugs"},"coverage":"reviewed","gaps":["A concrete evaluation still needs a protocol version, dataset release, eligible/scored counts and documented uncertainty. This guide does not manufacture those run-specific values."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed named primary documentation and existing protocol records. Editorial task scope is distinguished from source-reported procedures. No model execution or new numerical result; a reviewed guide is not a complete runnable protocol."}}}} {"id":"catalog-task-ligand-pose","kind":"benchmark","name":"Protein–ligand pose","description":"","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":["catalog-source-chai-1","catalog-source-boltz-2","catalog-source-diffdock-l","catalog-source-alphafold-3-server","catalog-source-vina"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"Protein–ligand pose","scope_note":"Predict the bound ligand geometry from prepared molecular inputs.","historical_missing_metadata":{"protocol_version":"not_yet_extracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"Protein–ligand pose evaluation checks both agreement with a reference pose and the plausibility of the predicted geometry.","summary_source_ids":["src-discovery-maabuu-posebusters","src-discovery-plinder-org-plinder"],"summary_source_locator":"PoseBusters README: Usage; PLINDER README: Gold standard benchmark sets and Plinder versions","sections":[{"title":"Choosing an evaluation","body":"Record whether a method receives an experimental bound protein, an unbound structure or a predicted receptor. PLINDER supplies related systems and versioned splits; PoseBusters supplies plausibility checks. These resources have different roles. Keep geometric validity separate from pose recovery and state any filtering of failed predictions.","source_ids":["src-discovery-maabuu-posebusters","src-discovery-plinder-org-plinder"],"source_locator":"PoseBusters README: Usage; PLINDER README: Gold standard benchmark sets and Plinder versions"},{"title":"Task scope","body":"This is a task guide, not a single versioned benchmark protocol. The connected resources provide examples or concrete procedures. A candidate method or proposed control is not evidence that an evaluation has been completed.","source_ids":["src-discovery-maabuu-posebusters","src-discovery-plinder-org-plinder"],"source_locator":"PoseBusters README: Usage; PLINDER README: Gold standard benchmark sets and Plinder versions"}],"facts":[{"label":"Entity type","value":"Task guide; concrete protocol identities remain separate.","status":"source_checked","source_ids":["src-discovery-maabuu-posebusters","src-discovery-plinder-org-plinder"],"source_locator":"PoseBusters README: Usage; PLINDER README: Gold standard benchmark sets and Plinder versions"},{"label":"Datasets","value":"No single dataset is fixed by this guide. Select a linked protocol and its versioned data release.","status":"inapplicable","source_ids":["src-discovery-maabuu-posebusters","src-discovery-plinder-org-plinder"],"source_locator":"PoseBusters README: Usage; PLINDER README: Gold standard benchmark sets and Plinder versions"},{"label":"Organisms","value":"No shared organism population is defined at this guide level. Record it for each selected dataset.","status":"inapplicable","source_ids":["src-discovery-maabuu-posebusters","src-discovery-plinder-org-plinder"],"source_locator":"PoseBusters README: Usage; PLINDER README: Gold standard benchmark sets and Plinder versions"},{"label":"Assays","value":"No single measurement assay is fixed by this guide; the endpoint and assay belong to the selected protocol.","status":"inapplicable","source_ids":["src-discovery-maabuu-posebusters","src-discovery-plinder-org-plinder"],"source_locator":"PoseBusters README: Usage; PLINDER README: Gold standard benchmark sets and Plinder versions"},{"label":"Splits","value":"No executable split is attached to this task identity. Use the selected protocol’s split manifest.","status":"inapplicable","source_ids":["src-discovery-maabuu-posebusters","src-discovery-plinder-org-plinder"],"source_locator":"PoseBusters README: Usage; PLINDER README: Gold standard benchmark sets and Plinder versions"},{"label":"Allowed inputs","value":"Predicted ligand coordinates, receptor information allowed by the protocol, and an experimental reference pose where pose recovery is scored.","status":"source_checked","source_ids":["src-discovery-maabuu-posebusters","src-discovery-plinder-org-plinder"],"source_locator":"PoseBusters README: Usage; PLINDER README: Gold standard benchmark sets and Plinder versions"},{"label":"Adaptation","value":"No common fitting regime is imposed here. Keep pretrained, frozen, probed, fine-tuned and conventional methods distinct where applicable.","status":"inapplicable","source_ids":["src-discovery-maabuu-posebusters","src-discovery-plinder-org-plinder"],"source_locator":"PoseBusters README: Usage; PLINDER README: Gold standard benchmark sets and Plinder versions"},{"label":"Metrics","value":"Reference-pose error and explicitly versioned plausibility checks. A validity pass without a matching pose is a different outcome from correct pose recovery.","status":"source_checked","source_ids":["src-discovery-maabuu-posebusters","src-discovery-plinder-org-plinder"],"source_locator":"PoseBusters README: Usage; PLINDER README: Gold standard benchmark sets and Plinder versions"},{"label":"Baselines","value":"Use classical docking or learned comparators only under matched receptor and pocket information. Candidate applicability does not establish a completed run.","status":"source_checked","source_ids":["src-discovery-maabuu-posebusters","src-discovery-plinder-org-plinder"],"source_locator":"PoseBusters README: Usage; PLINDER README: Gold standard benchmark sets and Plinder versions"}],"strengths":[{"text":"Plausibility checks catch physically problematic outputs that a geometric alignment score alone may miss.","source_ids":["src-discovery-maabuu-posebusters","src-discovery-plinder-org-plinder"],"source_locator":"PoseBusters README: Usage; PLINDER README: Gold standard benchmark sets and Plinder versions"}],"limitations":[{"text":"Results depend on receptor state, reference quality and failure handling. Pose accuracy does not establish binding affinity.","source_ids":["src-discovery-maabuu-posebusters","src-discovery-plinder-org-plinder"],"source_locator":"PoseBusters README: Usage; PLINDER README: Gold standard benchmark sets and Plinder versions"}],"diagram":{"title":"Conceptual evaluation workflow","steps":["Choose receptor state and split","Predict ligand pose","Check reference-pose agreement","Check geometry and report failures"],"caption":"Conceptual task guide. Dataset preparation, parameters and scoring must come from a separately identified protocol.","source_ids":["src-discovery-maabuu-posebusters","src-discovery-plinder-org-plinder"],"source_locator":"PoseBusters README: Usage; PLINDER README: Gold standard benchmark sets and Plinder versions"},"coverage":"reviewed","gaps":["A concrete evaluation still needs a protocol version, dataset release, eligible/scored counts and documented uncertainty. This guide does not manufacture those run-specific values."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed named primary documentation and existing protocol records. Editorial task scope is distinguished from source-reported procedures. No model execution or new numerical result; a reviewed guide is not a complete runnable protocol."}}}} {"id":"catalog-task-long-range-regulation","kind":"benchmark","name":"Long-range regulation","description":"","status":"discovered","facets":{"areas":["dna-genomes"]},"source_ids":["catalog-source-evo-2","catalog-source-alphagenome"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"Long-range regulation","scope_note":"Predict gene-expression or chromatin effects from long-context DNA sequence.","historical_missing_metadata":{"protocol_version":"not_yet_extracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"Long-range regulatory evaluation asks whether useful predictions depend on information beyond a short local DNA window.","summary_source_ids":["src-discovery-frederikkemarin-bend","dart-eval-regulatory-2024"],"summary_source_locator":"BEND README: Data format and training/evaluating supervised models; DART-Eval paper: task and context distinctions","sections":[{"title":"Choosing an evaluation","body":"A large supported input window is a model capability, not by itself a test of distant regulation. Select a concrete regulatory endpoint and retain genomic coordinates, split membership and permitted context. BEND’s coordinate-based task format illustrates why those details matter. Compare configurations only when their prediction target and available information are explicit.","source_ids":["src-discovery-frederikkemarin-bend","dart-eval-regulatory-2024"],"source_locator":"BEND README: Data format and training/evaluating supervised models; DART-Eval paper: task and context distinctions"},{"title":"Task scope","body":"This is a task guide, not a single versioned benchmark protocol. The connected resources provide examples or concrete procedures. A candidate method or proposed control is not evidence that an evaluation has been completed.","source_ids":["src-discovery-frederikkemarin-bend","dart-eval-regulatory-2024"],"source_locator":"BEND README: Data format and training/evaluating supervised models; DART-Eval paper: task and context distinctions"}],"facts":[{"label":"Entity type","value":"Task guide; concrete protocol identities remain separate.","status":"source_checked","source_ids":["src-discovery-frederikkemarin-bend","dart-eval-regulatory-2024"],"source_locator":"BEND README: Data format and training/evaluating supervised models; DART-Eval paper: task and context distinctions"},{"label":"Datasets","value":"No single dataset is fixed by this guide. Select a linked protocol and its versioned data release.","status":"inapplicable","source_ids":["src-discovery-frederikkemarin-bend","dart-eval-regulatory-2024"],"source_locator":"BEND README: Data format and training/evaluating supervised models; DART-Eval paper: task and context distinctions"},{"label":"Organisms","value":"No shared organism population is defined at this guide level. Record it for each selected dataset.","status":"inapplicable","source_ids":["src-discovery-frederikkemarin-bend","dart-eval-regulatory-2024"],"source_locator":"BEND README: Data format and training/evaluating supervised models; DART-Eval paper: task and context distinctions"},{"label":"Assays","value":"No single measurement assay is fixed by this guide; the endpoint and assay belong to the selected protocol.","status":"inapplicable","source_ids":["src-discovery-frederikkemarin-bend","dart-eval-regulatory-2024"],"source_locator":"BEND README: Data format and training/evaluating supervised models; DART-Eval paper: task and context distinctions"},{"label":"Splits","value":"No executable split is attached to this task identity. Use the selected protocol’s split manifest.","status":"inapplicable","source_ids":["src-discovery-frederikkemarin-bend","dart-eval-regulatory-2024"],"source_locator":"BEND README: Data format and training/evaluating supervised models; DART-Eval paper: task and context distinctions"},{"label":"Allowed inputs","value":"Genomic sequences with coordinates, a declared context window and the labels of a specific regulatory task.","status":"source_checked","source_ids":["src-discovery-frederikkemarin-bend","dart-eval-regulatory-2024"],"source_locator":"BEND README: Data format and training/evaluating supervised models; DART-Eval paper: task and context distinctions"},{"label":"Adaptation","value":"No common fitting regime is imposed here. Keep pretrained, frozen, probed, fine-tuned and conventional methods distinct where applicable.","status":"inapplicable","source_ids":["src-discovery-frederikkemarin-bend","dart-eval-regulatory-2024"],"source_locator":"BEND README: Data format and training/evaluating supervised models; DART-Eval paper: task and context distinctions"},{"label":"Metrics","value":"Use the concrete task’s metric and aggregation; there is no single score for all long-range regulation.","status":"source_checked","source_ids":["src-discovery-frederikkemarin-bend","dart-eval-regulatory-2024"],"source_locator":"BEND README: Data format and training/evaluating supervised models; DART-Eval paper: task and context distinctions"},{"label":"Baselines","value":"Matched short-context or task-specific methods can be proposed as controls, but require an explicit protocol and measured results before comparison.","status":"source_checked","source_ids":["src-discovery-frederikkemarin-bend","dart-eval-regulatory-2024"],"source_locator":"BEND README: Data format and training/evaluating supervised models; DART-Eval paper: task and context distinctions"}],"strengths":[{"text":"Recording context and coordinates makes the intended information advantage inspectable.","source_ids":["src-discovery-frederikkemarin-bend","dart-eval-regulatory-2024"],"source_locator":"BEND README: Data format and training/evaluating supervised models; DART-Eval paper: task and context distinctions"}],"limitations":[{"text":"This guide does not establish that longer context improves a model, nor does it provide a shared enhancer–gene or contact-map protocol.","source_ids":["src-discovery-frederikkemarin-bend","dart-eval-regulatory-2024"],"source_locator":"BEND README: Data format and training/evaluating supervised models; DART-Eval paper: task and context distinctions"}],"diagram":{"title":"Conceptual evaluation workflow","steps":["Define endpoint and genomic split","Declare local and distant information","Evaluate specified configurations","Interpret scores within that endpoint"],"caption":"Conceptual task guide. Dataset preparation, parameters and scoring must come from a separately identified protocol.","source_ids":["src-discovery-frederikkemarin-bend","dart-eval-regulatory-2024"],"source_locator":"BEND README: Data format and training/evaluating supervised models; DART-Eval paper: task and context distinctions"},"coverage":"reviewed","gaps":["A concrete evaluation still needs a protocol version, dataset release, eligible/scored counts and documented uncertainty. This guide does not manufacture those run-specific values."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed named primary documentation and existing protocol records. Editorial task scope is distinguished from source-reported procedures. No model execution or new numerical result; a reviewed guide is not a complete runnable protocol."}}}} {"id":"catalog-task-mfass-splice","kind":"benchmark","name":"MFASS splice-variant prioritisation","description":"","status":"discovered","facets":{"areas":["dna-genomes"]},"source_ids":["catalog-source-dnabert-2","catalog-source-nt-v2","catalog-source-spliceai","catalog-source-pangolin"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"MFASS splice-variant prioritisation","scope_note":"Functional exon-recognition assay; mfass-v2 reports a corrected baseline and one complete local DNABERT-2 protocol.","historical_missing_metadata":{"protocol_version":"not_yet_extracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"MFASS prioritisation asks whether variant scores enrich for experimentally disrupted exon recognition.","summary_source_ids":["rewire-mfass-v2-source"],"summary_source_locator":"Pinned MFASS v2 README: Correction, Dataset, Cohort reconciliation and Split; committed results files","sections":[{"title":"Choosing an evaluation","body":"The corrected rewire v2 protocol validates assay-oriented reference and mutant pairs and uses a fixed grouped split. Its baseline window is placed around the validated variant position. This task guide links that concrete protocol without replacing its identity. Preserve specialist sequence context, missing predictions and the chosen review capacity when comparing methods.","source_ids":["rewire-mfass-v2-source"],"source_locator":"Pinned MFASS v2 README: Correction, Dataset, Cohort reconciliation and Split; committed results files"},{"title":"Task scope","body":"This is a task guide, not a single versioned benchmark protocol. The connected resources provide examples or concrete procedures. A candidate method or proposed control is not evidence that an evaluation has been completed.","source_ids":["rewire-mfass-v2-source"],"source_locator":"Pinned MFASS v2 README: Correction, Dataset, Cohort reconciliation and Split; committed results files"}],"facts":[{"label":"Entity type","value":"Task guide; concrete protocol identities remain separate.","status":"source_checked","source_ids":["rewire-mfass-v2-source"],"source_locator":"Pinned MFASS v2 README: Correction, Dataset, Cohort reconciliation and Split; committed results files"},{"label":"Datasets","value":"The linked rewire MFASS v2 protocol defines its reconciled eligible variants and fixed test cohort; this guide is not another dataset release.","status":"source_checked","source_ids":["rewire-mfass-v2-source"],"source_locator":"Pinned MFASS v2 README: Correction, Dataset, Cohort reconciliation and Split; committed results files"},{"label":"Organisms","value":"Human variants evaluated through the MFASS reporter assay; this is not a population-level clinical validation.","status":"source_checked","source_ids":["rewire-mfass-v2-source"],"source_locator":"Pinned MFASS v2 README: Correction, Dataset, Cohort reconciliation and Split; committed results files"},{"label":"Assays","value":"MFASS reporter-based exon-recognition measurements, using the labels and eligibility rules preserved by the v2 protocol.","status":"source_checked","source_ids":["rewire-mfass-v2-source"],"source_locator":"Pinned MFASS v2 README: Correction, Dataset, Cohort reconciliation and Split; committed results files"},{"label":"Splits","value":"Use the grouped split-v2 manifests in the pinned runner revision. Preserve the train/test assignment and exclusions.","status":"source_checked","source_ids":["rewire-mfass-v2-source"],"source_locator":"Pinned MFASS v2 README: Correction, Dataset, Cohort reconciliation and Split; committed results files"},{"label":"Allowed inputs","value":"Variant scores aligned to the MFASS reporter-assay labels; assay-oriented sequence pairs for the corrected local baseline.","status":"source_checked","source_ids":["rewire-mfass-v2-source"],"source_locator":"Pinned MFASS v2 README: Correction, Dataset, Cohort reconciliation and Split; committed results files"},{"label":"Adaptation","value":"The corrected baseline and frozen-encoder logistic pipeline fit the training arm; the specialist scorers retain their published configurations.","status":"source_checked","source_ids":["rewire-mfass-v2-source"],"source_locator":"Pinned MFASS v2 README: Correction, Dataset, Cohort reconciliation and Split; committed results files"},{"label":"Metrics","value":"Precision at 100, average precision and AUROC in the pinned v2 evaluation. Coverage and paired uncertainty belong beside each comparison.","status":"source_checked","source_ids":["rewire-mfass-v2-source"],"source_locator":"Pinned MFASS v2 README: Correction, Dataset, Cohort reconciliation and Split; committed results files"},{"label":"Baselines","value":"The v2 k-mer/position baseline is a trained comparator. SpliceAI, Pangolin and the frozen DNABERT-2 logistic pipeline retain their own input and fitting definitions.","status":"source_checked","source_ids":["rewire-mfass-v2-source"],"source_locator":"Pinned MFASS v2 README: Correction, Dataset, Cohort reconciliation and Split; committed results files"}],"strengths":[{"text":"Functional reporter labels provide an assay endpoint separate from clinical assertions.","source_ids":["rewire-mfass-v2-source"],"source_locator":"Pinned MFASS v2 README: Correction, Dataset, Cohort reconciliation and Split; committed results files"}],"limitations":[{"text":"Reporter effects are not clinical diagnoses. Different sequence context and missing-prediction coverage prevent an identical-input interpretation.","source_ids":["rewire-mfass-v2-source"],"source_locator":"Pinned MFASS v2 README: Correction, Dataset, Cohort reconciliation and Split; committed results files"}],"diagram":{"title":"Conceptual evaluation workflow","steps":["Validate assay-oriented sequence pairs","Apply the fixed grouped split","Score each declared method","Report prioritisation, ranking and coverage"],"caption":"Conceptual task guide. Dataset preparation, parameters and scoring must come from a separately identified protocol.","source_ids":["rewire-mfass-v2-source"],"source_locator":"Pinned MFASS v2 README: Correction, Dataset, Cohort reconciliation and Split; committed results files"},"coverage":"reviewed","gaps":["A concrete evaluation still needs a protocol version, dataset release, eligible/scored counts and documented uncertainty. This guide does not manufacture those run-specific values."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed named primary documentation and existing protocol records. Editorial task scope is distinguished from source-reported procedures. No model execution or new numerical result; a reviewed guide is not a complete runnable protocol."}}}} {"id":"catalog-task-microbial-promoters","kind":"benchmark","name":"Bacterial promoter prediction","description":"","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":["catalog-source-prokbert","catalog-source-evo-2"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"Bacterial promoter prediction","scope_note":"Classify promoter activity from microbial DNA sequence.","historical_missing_metadata":{"protocol_version":"not_yet_extracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"Bacterial promoter prediction evaluates whether a sequence model can distinguish promoter-labelled sequences under a specified dataset definition.","summary_source_ids":["evidence-official-2efbfaba5a1c09f4f7fa"],"summary_source_locator":"ProkBERT README: promoter prediction task and fine-tuning examples","sections":[{"title":"Choosing an evaluation","body":"A concrete study must define what counts as a promoter, how negative examples are selected and which organisms or sequence groups are held out. Keep those decisions with the score. ProkBERT documents a promoter-prediction use case, but that implementation does not make every promoter dataset or split interchangeable.","source_ids":["evidence-official-2efbfaba5a1c09f4f7fa"],"source_locator":"ProkBERT README: promoter prediction task and fine-tuning examples"},{"title":"Task scope","body":"This is a task guide, not a single versioned benchmark protocol. The connected resources provide examples or concrete procedures. A candidate method or proposed control is not evidence that an evaluation has been completed.","source_ids":["evidence-official-2efbfaba5a1c09f4f7fa"],"source_locator":"ProkBERT README: promoter prediction task and fine-tuning examples"}],"facts":[{"label":"Entity type","value":"Task guide; concrete protocol identities remain separate.","status":"source_checked","source_ids":["evidence-official-2efbfaba5a1c09f4f7fa"],"source_locator":"ProkBERT README: promoter prediction task and fine-tuning examples"},{"label":"Datasets","value":"No single dataset is fixed by this guide. Select a linked protocol and its versioned data release.","status":"inapplicable","source_ids":["evidence-official-2efbfaba5a1c09f4f7fa"],"source_locator":"ProkBERT README: promoter prediction task and fine-tuning examples"},{"label":"Organisms","value":"No shared organism population is defined at this guide level. Record it for each selected dataset.","status":"inapplicable","source_ids":["evidence-official-2efbfaba5a1c09f4f7fa"],"source_locator":"ProkBERT README: promoter prediction task and fine-tuning examples"},{"label":"Assays","value":"No single measurement assay is fixed by this guide; the endpoint and assay belong to the selected protocol.","status":"inapplicable","source_ids":["evidence-official-2efbfaba5a1c09f4f7fa"],"source_locator":"ProkBERT README: promoter prediction task and fine-tuning examples"},{"label":"Splits","value":"No executable split is attached to this task identity. Use the selected protocol’s split manifest.","status":"inapplicable","source_ids":["evidence-official-2efbfaba5a1c09f4f7fa"],"source_locator":"ProkBERT README: promoter prediction task and fine-tuning examples"},{"label":"Allowed inputs","value":"DNA sequences and promoter labels from a named reference dataset; organism and negative-set construction are protocol-specific.","status":"source_checked","source_ids":["evidence-official-2efbfaba5a1c09f4f7fa"],"source_locator":"ProkBERT README: promoter prediction task and fine-tuning examples"},{"label":"Adaptation","value":"No common fitting regime is imposed here. Keep pretrained, frozen, probed, fine-tuned and conventional methods distinct where applicable.","status":"inapplicable","source_ids":["evidence-official-2efbfaba5a1c09f4f7fa"],"source_locator":"ProkBERT README: promoter prediction task and fine-tuning examples"},{"label":"Metrics","value":"Classification metrics defined by the selected dataset and protocol; decision thresholds and class balance must be reported.","status":"source_checked","source_ids":["evidence-official-2efbfaba5a1c09f4f7fa"],"source_locator":"ProkBERT README: promoter prediction task and fine-tuning examples"},{"label":"Baselines","value":"Documented sequence classifiers or simple composition controls may be candidates; no measured baseline value is supplied by this guide.","status":"source_checked","source_ids":["evidence-official-2efbfaba5a1c09f4f7fa"],"source_locator":"ProkBERT README: promoter prediction task and fine-tuning examples"}],"strengths":[{"text":"An explicit promoter-label task gives a sequence representation a measurable downstream endpoint.","source_ids":["evidence-official-2efbfaba5a1c09f4f7fa"],"source_locator":"ProkBERT README: promoter prediction task and fine-tuning examples"}],"limitations":[{"text":"A classifier can exploit how negatives were sampled. Transfer to another organism or promoter definition needs its own evaluation.","source_ids":["evidence-official-2efbfaba5a1c09f4f7fa"],"source_locator":"ProkBERT README: promoter prediction task and fine-tuning examples"}],"diagram":{"title":"Conceptual evaluation workflow","steps":["Choose promoter labels and negatives","Fix organism or sequence holdout","Evaluate the defined classifier","Report classification and coverage"],"caption":"Conceptual task guide. Dataset preparation, parameters and scoring must come from a separately identified protocol.","source_ids":["evidence-official-2efbfaba5a1c09f4f7fa"],"source_locator":"ProkBERT README: promoter prediction task and fine-tuning examples"},"coverage":"reviewed","gaps":["A concrete evaluation still needs a protocol version, dataset release, eligible/scored counts and documented uncertainty. This guide does not manufacture those run-specific values."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed named primary documentation and existing protocol records. Editorial task scope is distinguished from source-reported procedures. No model execution or new numerical result; a reviewed guide is not a complete runnable protocol."}}}} {"id":"catalog-task-phage-pathogen-reads","kind":"benchmark","name":"Phage / pathogen reads","description":"","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":["catalog-source-prokbert","catalog-source-metagene-1","catalog-source-kraken2"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"Phage / pathogen reads","scope_note":"Classify held-out phage or pathogen sequences and record taxonomic distance.","historical_missing_metadata":{"protocol_version":"not_yet_extracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"This task area covers microbial sequence-read classification. Identifying a phage and identifying a pathogen are distinct endpoints.","summary_source_ids":["evidence-official-2efbfaba5a1c09f4f7fa"],"summary_source_locator":"ProkBERT README: phage identification and downstream tasks. The combined task label is catalogue scope, not a claim that phage identity establishes pathogenicity.","sections":[{"title":"Choosing an evaluation","body":"Choose a concrete label definition before evaluating read classifications. A phage-versus-non-phage benchmark does not establish clinical pathogenicity or organism abundance. Record reference-library coverage and the taxonomic or sequence separation between training and assessment. This guide describes evaluation scope and introduces no sequence-design procedure.","source_ids":["evidence-official-2efbfaba5a1c09f4f7fa"],"source_locator":"ProkBERT README: phage identification and downstream tasks. The combined task label is catalogue scope, not a claim that phage identity establishes pathogenicity."},{"title":"Task scope","body":"This is a task guide, not a single versioned benchmark protocol. The connected resources provide examples or concrete procedures. A candidate method or proposed control is not evidence that an evaluation has been completed.","source_ids":["evidence-official-2efbfaba5a1c09f4f7fa"],"source_locator":"ProkBERT README: phage identification and downstream tasks. The combined task label is catalogue scope, not a claim that phage identity establishes pathogenicity."}],"facts":[{"label":"Entity type","value":"Task guide; concrete protocol identities remain separate.","status":"source_checked","source_ids":["evidence-official-2efbfaba5a1c09f4f7fa"],"source_locator":"ProkBERT README: phage identification and downstream tasks. The combined task label is catalogue scope, not a claim that phage identity establishes pathogenicity."},{"label":"Datasets","value":"No single dataset is fixed by this guide. Select a linked protocol and its versioned data release.","status":"inapplicable","source_ids":["evidence-official-2efbfaba5a1c09f4f7fa"],"source_locator":"ProkBERT README: phage identification and downstream tasks. The combined task label is catalogue scope, not a claim that phage identity establishes pathogenicity."},{"label":"Organisms","value":"No shared organism population is defined at this guide level. Record it for each selected dataset.","status":"inapplicable","source_ids":["evidence-official-2efbfaba5a1c09f4f7fa"],"source_locator":"ProkBERT README: phage identification and downstream tasks. The combined task label is catalogue scope, not a claim that phage identity establishes pathogenicity."},{"label":"Assays","value":"No single measurement assay is fixed by this guide; the endpoint and assay belong to the selected protocol.","status":"inapplicable","source_ids":["evidence-official-2efbfaba5a1c09f4f7fa"],"source_locator":"ProkBERT README: phage identification and downstream tasks. The combined task label is catalogue scope, not a claim that phage identity establishes pathogenicity."},{"label":"Splits","value":"No executable split is attached to this task identity. Use the selected protocol’s split manifest.","status":"inapplicable","source_ids":["evidence-official-2efbfaba5a1c09f4f7fa"],"source_locator":"ProkBERT README: phage identification and downstream tasks. The combined task label is catalogue scope, not a claim that phage identity establishes pathogenicity."},{"label":"Allowed inputs","value":"Sequence reads or fragments with task-specific reference labels; reference database identity and version are part of the protocol.","status":"source_checked","source_ids":["evidence-official-2efbfaba5a1c09f4f7fa"],"source_locator":"ProkBERT README: phage identification and downstream tasks. The combined task label is catalogue scope, not a claim that phage identity establishes pathogenicity."},{"label":"Adaptation","value":"No common fitting regime is imposed here. Keep pretrained, frozen, probed, fine-tuned and conventional methods distinct where applicable.","status":"inapplicable","source_ids":["evidence-official-2efbfaba5a1c09f4f7fa"],"source_locator":"ProkBERT README: phage identification and downstream tasks. The combined task label is catalogue scope, not a claim that phage identity establishes pathogenicity."},{"label":"Metrics","value":"Label-specific classification metrics and explicit false-positive/negative counts. State the unit scored and the treatment of unclassified reads.","status":"source_checked","source_ids":["evidence-official-2efbfaba5a1c09f4f7fa"],"source_locator":"ProkBERT README: phage identification and downstream tasks. The combined task label is catalogue scope, not a claim that phage identity establishes pathogenicity."},{"label":"Baselines","value":"Compare against the conventional classifier specified by the selected protocol using the same reference information; candidate tools alone are not evidence.","status":"source_checked","source_ids":["evidence-official-2efbfaba5a1c09f4f7fa"],"source_locator":"ProkBERT README: phage identification and downstream tasks. The combined task label is catalogue scope, not a claim that phage identity establishes pathogenicity."}],"strengths":[{"text":"A defined label and held-out reference relationship make sequence-classification claims testable.","source_ids":["evidence-official-2efbfaba5a1c09f4f7fa"],"source_locator":"ProkBERT README: phage identification and downstream tasks. The combined task label is catalogue scope, not a claim that phage identity establishes pathogenicity."}],"limitations":[{"text":"Do not use a phage-detection score as evidence of pathogenicity, diagnostic accuracy or community abundance.","source_ids":["evidence-official-2efbfaba5a1c09f4f7fa"],"source_locator":"ProkBERT README: phage identification and downstream tasks. The combined task label is catalogue scope, not a claim that phage identity establishes pathogenicity."}],"diagram":{"title":"Conceptual evaluation workflow","steps":["Define the classification endpoint","Document reference and holdout scope","Classify held-out reads","Report label-specific errors"],"caption":"Conceptual task guide. Dataset preparation, parameters and scoring must come from a separately identified protocol.","source_ids":["evidence-official-2efbfaba5a1c09f4f7fa"],"source_locator":"ProkBERT README: phage identification and downstream tasks. The combined task label is catalogue scope, not a claim that phage identity establishes pathogenicity."},"coverage":"reviewed","gaps":["A concrete evaluation still needs a protocol version, dataset release, eligible/scored counts and documented uncertainty. This guide does not manufacture those run-specific values."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed named primary documentation and existing protocol records. Editorial task scope is distinguished from source-reported procedures. No model execution or new numerical result; a reviewed guide is not a complete runnable protocol."}}}} {"id":"catalog-task-protein-design","kind":"benchmark","name":"Protein design / inverse folding","description":"","status":"discovered","facets":{"areas":["proteins-complexes"]},"source_ids":["catalog-source-proteinmpnn"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"Protein design / inverse folding","scope_note":"Score or design sequences conditional on a known structure.","historical_missing_metadata":{"protocol_version":"not_yet_extracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"Inverse-folding evaluation asks whether a method can propose amino-acid sequences compatible with a specified protein structure.","summary_source_ids":["src-discovery-dauparas-proteinmpnn"],"summary_source_locator":"ProteinMPNN README: overview, scoring outputs and training documentation","sections":[{"title":"Choosing an evaluation","body":"Keep inverse folding separate from sequence-only generation and experimental function. ProteinMPNN takes structural context and produces sequence probabilities or candidate sequences. A concrete benchmark must state the held-out structures, permitted constraints and assessment. Recovery of a reference sequence and experimental success answer different questions.","source_ids":["src-discovery-dauparas-proteinmpnn"],"source_locator":"ProteinMPNN README: overview, scoring outputs and training documentation"},{"title":"Task scope","body":"This is a task guide, not a single versioned benchmark protocol. The connected resources provide examples or concrete procedures. A candidate method or proposed control is not evidence that an evaluation has been completed.","source_ids":["src-discovery-dauparas-proteinmpnn"],"source_locator":"ProteinMPNN README: overview, scoring outputs and training documentation"}],"facts":[{"label":"Entity type","value":"Task guide; concrete protocol identities remain separate.","status":"source_checked","source_ids":["src-discovery-dauparas-proteinmpnn"],"source_locator":"ProteinMPNN README: overview, scoring outputs and training documentation"},{"label":"Datasets","value":"No single dataset is fixed by this guide. Select a linked protocol and its versioned data release.","status":"inapplicable","source_ids":["src-discovery-dauparas-proteinmpnn"],"source_locator":"ProteinMPNN README: overview, scoring outputs and training documentation"},{"label":"Organisms","value":"No shared organism population is defined at this guide level. Record it for each selected dataset.","status":"inapplicable","source_ids":["src-discovery-dauparas-proteinmpnn"],"source_locator":"ProteinMPNN README: overview, scoring outputs and training documentation"},{"label":"Assays","value":"No single measurement assay is fixed by this guide; the endpoint and assay belong to the selected protocol.","status":"inapplicable","source_ids":["src-discovery-dauparas-proteinmpnn"],"source_locator":"ProteinMPNN README: overview, scoring outputs and training documentation"},{"label":"Splits","value":"No executable split is attached to this task identity. Use the selected protocol’s split manifest.","status":"inapplicable","source_ids":["src-discovery-dauparas-proteinmpnn"],"source_locator":"ProteinMPNN README: overview, scoring outputs and training documentation"},{"label":"Allowed inputs","value":"A target protein backbone and declared structural or sequence constraints for an inverse-folding protocol.","status":"source_checked","source_ids":["src-discovery-dauparas-proteinmpnn"],"source_locator":"ProteinMPNN README: overview, scoring outputs and training documentation"},{"label":"Adaptation","value":"No common fitting regime is imposed here. Keep pretrained, frozen, probed, fine-tuned and conventional methods distinct where applicable.","status":"inapplicable","source_ids":["src-discovery-dauparas-proteinmpnn"],"source_locator":"ProteinMPNN README: overview, scoring outputs and training documentation"},{"label":"Metrics","value":"Protocol-specific sequence recovery or structural assessment; experimental validation, when available, is a separate endpoint.","status":"source_checked","source_ids":["src-discovery-dauparas-proteinmpnn"],"source_locator":"ProteinMPNN README: overview, scoring outputs and training documentation"},{"label":"Baselines","value":"Use inverse-folding comparators supplied with the selected benchmark and match structural inputs and constraints.","status":"source_checked","source_ids":["src-discovery-dauparas-proteinmpnn"],"source_locator":"ProteinMPNN README: overview, scoring outputs and training documentation"}],"strengths":[{"text":"A specified structural target provides a clear conditioning context for evaluating sequence proposals.","source_ids":["src-discovery-dauparas-proteinmpnn"],"source_locator":"ProteinMPNN README: overview, scoring outputs and training documentation"}],"limitations":[{"text":"Reference-sequence recovery does not by itself establish folding, function or experimental success.","source_ids":["src-discovery-dauparas-proteinmpnn"],"source_locator":"ProteinMPNN README: overview, scoring outputs and training documentation"}],"diagram":{"title":"Conceptual evaluation workflow","steps":["Select held-out structural targets","Declare allowed conditioning","Generate or score candidate sequences","Evaluate the specified endpoint"],"caption":"Conceptual task guide. Dataset preparation, parameters and scoring must come from a separately identified protocol.","source_ids":["src-discovery-dauparas-proteinmpnn"],"source_locator":"ProteinMPNN README: overview, scoring outputs and training documentation"},"coverage":"reviewed","gaps":["A concrete evaluation still needs a protocol version, dataset release, eligible/scored counts and documented uncertainty. This guide does not manufacture those run-specific values."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed named primary documentation and existing protocol records. Editorial task scope is distinguished from source-reported procedures. No model execution or new numerical result; a reviewed guide is not a complete runnable protocol."}}}} {"id":"catalog-task-protein-monomer-structure","kind":"benchmark","name":"Monomer structure","description":"","status":"discovered","facets":{"areas":["proteins-complexes"]},"source_ids":["catalog-source-esm-2","catalog-source-esmfold","catalog-source-chai-1"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"Monomer structure","scope_note":"Predict single-chain structure from sequence.","historical_missing_metadata":{"protocol_version":"not_yet_extracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"Monomer-structure evaluation measures the accuracy of a predicted individual protein structure against a specified reference.","summary_source_ids":["evidence-alphafold-paper"],"summary_source_locator":"AlphaFold 3 Methods: Metrics and recent PDB comparisons. Monomer scope is defined by this catalogue guide.","sections":[{"title":"Choosing an evaluation","body":"Define whether the method receives only sequence or also evolutionary and template information. Score the individual chain with the chosen alignment and residue-coverage rules. A monomer score is not a protein-interface score, and confidence estimates are not experimental reference measurements.","source_ids":["evidence-alphafold-paper"],"source_locator":"AlphaFold 3 Methods: Metrics and recent PDB comparisons. Monomer scope is defined by this catalogue guide."},{"title":"Task scope","body":"This is a task guide, not a single versioned benchmark protocol. The connected resources provide examples or concrete procedures. A candidate method or proposed control is not evidence that an evaluation has been completed.","source_ids":["evidence-alphafold-paper"],"source_locator":"AlphaFold 3 Methods: Metrics and recent PDB comparisons. Monomer scope is defined by this catalogue guide."}],"facts":[{"label":"Entity type","value":"Task guide; concrete protocol identities remain separate.","status":"source_checked","source_ids":["evidence-alphafold-paper"],"source_locator":"AlphaFold 3 Methods: Metrics and recent PDB comparisons. Monomer scope is defined by this catalogue guide."},{"label":"Datasets","value":"No single dataset is fixed by this guide. Select a linked protocol and its versioned data release.","status":"inapplicable","source_ids":["evidence-alphafold-paper"],"source_locator":"AlphaFold 3 Methods: Metrics and recent PDB comparisons. Monomer scope is defined by this catalogue guide."},{"label":"Organisms","value":"No shared organism population is defined at this guide level. Record it for each selected dataset.","status":"inapplicable","source_ids":["evidence-alphafold-paper"],"source_locator":"AlphaFold 3 Methods: Metrics and recent PDB comparisons. Monomer scope is defined by this catalogue guide."},{"label":"Assays","value":"No single measurement assay is fixed by this guide; the endpoint and assay belong to the selected protocol.","status":"inapplicable","source_ids":["evidence-alphafold-paper"],"source_locator":"AlphaFold 3 Methods: Metrics and recent PDB comparisons. Monomer scope is defined by this catalogue guide."},{"label":"Splits","value":"No executable split is attached to this task identity. Use the selected protocol’s split manifest.","status":"inapplicable","source_ids":["evidence-alphafold-paper"],"source_locator":"AlphaFold 3 Methods: Metrics and recent PDB comparisons. Monomer scope is defined by this catalogue guide."},{"label":"Allowed inputs","value":"A protein sequence with any protocol-permitted alignments or templates; an experimental chain structure for reference-based scoring.","status":"source_checked","source_ids":["evidence-alphafold-paper"],"source_locator":"AlphaFold 3 Methods: Metrics and recent PDB comparisons. Monomer scope is defined by this catalogue guide."},{"label":"Adaptation","value":"No common fitting regime is imposed here. Keep pretrained, frozen, probed, fine-tuned and conventional methods distinct where applicable.","status":"inapplicable","source_ids":["evidence-alphafold-paper"],"source_locator":"AlphaFold 3 Methods: Metrics and recent PDB comparisons. Monomer scope is defined by this catalogue guide."},{"label":"Metrics","value":"Structure-quality measures such as LDDT or TM-score where specified by the protocol; exact residue inclusion and aggregation remain explicit.","status":"source_checked","source_ids":["evidence-alphafold-paper"],"source_locator":"AlphaFold 3 Methods: Metrics and recent PDB comparisons. Monomer scope is defined by this catalogue guide."},{"label":"Baselines","value":"Compare structure predictors under declared input information and sampling budgets rather than treating all sequence-to-structure runs as equivalent.","status":"source_checked","source_ids":["evidence-alphafold-paper"],"source_locator":"AlphaFold 3 Methods: Metrics and recent PDB comparisons. Monomer scope is defined by this catalogue guide."}],"strengths":[{"text":"Chain-level reference comparison isolates a clearly defined structural endpoint.","source_ids":["evidence-alphafold-paper"],"source_locator":"AlphaFold 3 Methods: Metrics and recent PDB comparisons. Monomer scope is defined by this catalogue guide."}],"limitations":[{"text":"A correct monomer fold does not establish the arrangement or accuracy of a molecular complex.","source_ids":["evidence-alphafold-paper"],"source_locator":"AlphaFold 3 Methods: Metrics and recent PDB comparisons. Monomer scope is defined by this catalogue guide."}],"diagram":{"title":"Conceptual evaluation workflow","steps":["Fix reference chain and allowed inputs","Predict the monomer structure","Align and match scored residues","Measure declared structural accuracy"],"caption":"Conceptual task guide. Dataset preparation, parameters and scoring must come from a separately identified protocol.","source_ids":["evidence-alphafold-paper"],"source_locator":"AlphaFold 3 Methods: Metrics and recent PDB comparisons. Monomer scope is defined by this catalogue guide."},"coverage":"reviewed","gaps":["A concrete evaluation still needs a protocol version, dataset release, eligible/scored counts and documented uncertainty. This guide does not manufacture those run-specific values."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed named primary documentation and existing protocol records. Editorial task scope is distinguished from source-reported procedures. No model execution or new numerical result; a reviewed guide is not a complete runnable protocol."}}}} {"id":"catalog-task-proteingym-effects","kind":"benchmark","name":"ProteinGym mutation effects","description":"","status":"discovered","facets":{"areas":["proteins-complexes"]},"source_ids":["catalog-source-mimic","catalog-source-esm-2","catalog-source-proteinmpnn"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"ProteinGym mutation effects","scope_note":"Rank substitution effects within held-out deep-mutational-scanning assays.","historical_missing_metadata":{"protocol_version":"not_yet_extracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"ProteinGym mutation-effect evaluation compares variant scores with measurements from individual functional assays.","summary_source_ids":["src-discovery-oatml-markslab-proteingym"],"summary_source_locator":"README: Overview, Results, benchmark baselines and contribution notes on aggregation","sections":[{"title":"Choosing an evaluation","body":"Select the suite release, assay subset and zero-shot or supervised track. ProteinGym reports within-assay metrics and further aggregation by protein and functional category. Use the published aggregation rules; a simple mean across all assays is not automatically the suite’s reported score. Preserve the distinction between molecular assay effects and clinical labels.","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"README: Overview, Results, benchmark baselines and contribution notes on aggregation"},{"title":"Task scope","body":"This is a task guide, not a single versioned benchmark protocol. The connected resources provide examples or concrete procedures. A candidate method or proposed control is not evidence that an evaluation has been completed.","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"README: Overview, Results, benchmark baselines and contribution notes on aggregation"}],"facts":[{"label":"Entity type","value":"Task guide; concrete protocol identities remain separate.","status":"source_checked","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"README: Overview, Results, benchmark baselines and contribution notes on aggregation"},{"label":"Datasets","value":"Select a particular ProteinGym release, assay subset and evaluation track; the suite contains separate substitution and indel resources.","status":"source_checked","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"README: Overview, Results, benchmark baselines and contribution notes on aggregation"},{"label":"Organisms","value":"No shared organism population is defined at this guide level. Record it for each selected dataset.","status":"inapplicable","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"README: Overview, Results, benchmark baselines and contribution notes on aggregation"},{"label":"Assays","value":"Deep mutational scanning measurements for molecular effects; assay metadata and functional categories remain distinct.","status":"source_checked","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"README: Overview, Results, benchmark baselines and contribution notes on aggregation"},{"label":"Splits","value":"No executable split is attached to this task identity. Use the selected protocol’s split manifest.","status":"inapplicable","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"README: Overview, Results, benchmark baselines and contribution notes on aggregation"},{"label":"Allowed inputs","value":"Protein variants, permitted sequence or structure information, and the selected assay’s reference measurements.","status":"source_checked","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"README: Overview, Results, benchmark baselines and contribution notes on aggregation"},{"label":"Adaptation","value":"No common fitting regime is imposed here. Keep pretrained, frozen, probed, fine-tuned and conventional methods distinct where applicable.","status":"inapplicable","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"README: Overview, Results, benchmark baselines and contribution notes on aggregation"},{"label":"Metrics","value":"DMS zero-shot tracks include Spearman, NDCG, AUC, MCC and top-k recall; supervised tracks include Spearman and MSE. Choose the track-specific definition.","status":"source_checked","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"README: Overview, Results, benchmark baselines and contribution notes on aggregation"},{"label":"Baselines","value":"ProteinGym supplies single-sequence, alignment-based and other comparator scores; their extra information and supervision must remain visible.","status":"source_checked","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"README: Overview, Results, benchmark baselines and contribution notes on aggregation"}],"strengths":[{"text":"Per-assay reporting exposes how performance varies across proteins and measurement types.","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"README: Overview, Results, benchmark baselines and contribution notes on aggregation"}],"limitations":[{"text":"Assays measure different properties and have different coverage. A pooled ranking can hide those differences and depends on the aggregation rule.","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"README: Overview, Results, benchmark baselines and contribution notes on aggregation"}],"diagram":{"title":"Conceptual evaluation workflow","steps":["Pin release, track and assays","Score the eligible variants","Compute per-assay metrics","Apply the declared aggregation"],"caption":"Conceptual task guide. Dataset preparation, parameters and scoring must come from a separately identified protocol.","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"README: Overview, Results, benchmark baselines and contribution notes on aggregation"},"coverage":"reviewed","gaps":["A concrete evaluation still needs a protocol version, dataset release, eligible/scored counts and documented uncertainty. This guide does not manufacture those run-specific values."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed named primary documentation and existing protocol records. Editorial task scope is distinguished from source-reported procedures. No model execution or new numerical result; a reviewed guide is not a complete runnable protocol."}}}} {"id":"catalog-task-rna-secondary-structure","kind":"benchmark","name":"RNA secondary structure","description":"","status":"discovered","facets":{"areas":["rna-transcriptomes"]},"source_ids":["catalog-source-rna-fm","catalog-source-rhofold"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"RNA secondary structure","scope_note":"Compare predicted base pairs against held-out RNA structures.","historical_missing_metadata":{"protocol_version":"not_yet_extracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"RNA secondary-structure prediction evaluates proposed nucleotide pairing rather than complete three-dimensional geometry.","summary_source_ids":["src-discovery-terry-r123-rnabenchmark","src-discovery-ml4bio-rna-fm"],"summary_source_locator":"BEACON README: Tasks and Datasets; RNA-FM README: secondary structure prediction and foundation-model outputs","sections":[{"title":"Choosing an evaluation","body":"Use reference pairing labels and a declared scoring rule. BEACON includes a dedicated secondary-structure task, while RNA-FM supplies representations and pairing-related outputs used by downstream models. A representation, a pairing predictor and a complete benchmark are different records. Keep sequence-family overlap and reference-label construction visible.","source_ids":["src-discovery-terry-r123-rnabenchmark","src-discovery-ml4bio-rna-fm"],"source_locator":"BEACON README: Tasks and Datasets; RNA-FM README: secondary structure prediction and foundation-model outputs"},{"title":"Task scope","body":"This is a task guide, not a single versioned benchmark protocol. The connected resources provide examples or concrete procedures. A candidate method or proposed control is not evidence that an evaluation has been completed.","source_ids":["src-discovery-terry-r123-rnabenchmark","src-discovery-ml4bio-rna-fm"],"source_locator":"BEACON README: Tasks and Datasets; RNA-FM README: secondary structure prediction and foundation-model outputs"}],"facts":[{"label":"Entity type","value":"Task guide; concrete protocol identities remain separate.","status":"source_checked","source_ids":["src-discovery-terry-r123-rnabenchmark","src-discovery-ml4bio-rna-fm"],"source_locator":"BEACON README: Tasks and Datasets; RNA-FM README: secondary structure prediction and foundation-model outputs"},{"label":"Datasets","value":"No single dataset is fixed by this guide. Select a linked protocol and its versioned data release.","status":"inapplicable","source_ids":["src-discovery-terry-r123-rnabenchmark","src-discovery-ml4bio-rna-fm"],"source_locator":"BEACON README: Tasks and Datasets; RNA-FM README: secondary structure prediction and foundation-model outputs"},{"label":"Organisms","value":"No shared organism population is defined at this guide level. Record it for each selected dataset.","status":"inapplicable","source_ids":["src-discovery-terry-r123-rnabenchmark","src-discovery-ml4bio-rna-fm"],"source_locator":"BEACON README: Tasks and Datasets; RNA-FM README: secondary structure prediction and foundation-model outputs"},{"label":"Assays","value":"No single measurement assay is fixed by this guide; the endpoint and assay belong to the selected protocol.","status":"inapplicable","source_ids":["src-discovery-terry-r123-rnabenchmark","src-discovery-ml4bio-rna-fm"],"source_locator":"BEACON README: Tasks and Datasets; RNA-FM README: secondary structure prediction and foundation-model outputs"},{"label":"Splits","value":"No executable split is attached to this task identity. Use the selected protocol’s split manifest.","status":"inapplicable","source_ids":["src-discovery-terry-r123-rnabenchmark","src-discovery-ml4bio-rna-fm"],"source_locator":"BEACON README: Tasks and Datasets; RNA-FM README: secondary structure prediction and foundation-model outputs"},{"label":"Allowed inputs","value":"RNA sequence and any permitted auxiliary information; reference secondary-structure labels for scoring.","status":"source_checked","source_ids":["src-discovery-terry-r123-rnabenchmark","src-discovery-ml4bio-rna-fm"],"source_locator":"BEACON README: Tasks and Datasets; RNA-FM README: secondary structure prediction and foundation-model outputs"},{"label":"Adaptation","value":"No common fitting regime is imposed here. Keep pretrained, frozen, probed, fine-tuned and conventional methods distinct where applicable.","status":"inapplicable","source_ids":["src-discovery-terry-r123-rnabenchmark","src-discovery-ml4bio-rna-fm"],"source_locator":"BEACON README: Tasks and Datasets; RNA-FM README: secondary structure prediction and foundation-model outputs"},{"label":"Metrics","value":"Pairing or secondary-structure metrics from the selected protocol. This guide defines no common pseudoknot policy, matching tolerance or aggregation.","status":"source_checked","source_ids":["src-discovery-terry-r123-rnabenchmark","src-discovery-ml4bio-rna-fm"],"source_locator":"BEACON README: Tasks and Datasets; RNA-FM README: secondary structure prediction and foundation-model outputs"},{"label":"Baselines","value":"Compare the learned predictor with the conventional folding or learned baselines of the concrete task; do not infer a measured result from model availability.","status":"source_checked","source_ids":["src-discovery-terry-r123-rnabenchmark","src-discovery-ml4bio-rna-fm"],"source_locator":"BEACON README: Tasks and Datasets; RNA-FM README: secondary structure prediction and foundation-model outputs"}],"strengths":[{"text":"Pairing labels give a distinct structural endpoint that can be evaluated separately from full 3D prediction.","source_ids":["src-discovery-terry-r123-rnabenchmark","src-discovery-ml4bio-rna-fm"],"source_locator":"BEACON README: Tasks and Datasets; RNA-FM README: secondary structure prediction and foundation-model outputs"}],"limitations":[{"text":"Secondary-structure accuracy does not establish tertiary-structure accuracy, and close sequence relatives can weaken a claimed generalisation test.","source_ids":["src-discovery-terry-r123-rnabenchmark","src-discovery-ml4bio-rna-fm"],"source_locator":"BEACON README: Tasks and Datasets; RNA-FM README: secondary structure prediction and foundation-model outputs"}],"diagram":{"title":"Conceptual evaluation workflow","steps":["Select RNA references and split","Declare permitted sequence context","Predict pairing structure","Apply the declared matching and scoring"],"caption":"Conceptual task guide. Dataset preparation, parameters and scoring must come from a separately identified protocol.","source_ids":["src-discovery-terry-r123-rnabenchmark","src-discovery-ml4bio-rna-fm"],"source_locator":"BEACON README: Tasks and Datasets; RNA-FM README: secondary structure prediction and foundation-model outputs"},"coverage":"reviewed","gaps":["A concrete evaluation still needs a protocol version, dataset release, eligible/scored counts and documented uncertainty. This guide does not manufacture those run-specific values."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed named primary documentation and existing protocol records. Editorial task scope is distinguished from source-reported procedures. No model execution or new numerical result; a reviewed guide is not a complete runnable protocol."}}}} {"id":"catalog-task-rna-splice-sites","kind":"benchmark","name":"RNA splice-site mapping","description":"","status":"discovered","facets":{"areas":["rna-transcriptomes"]},"source_ids":["catalog-source-rna-fm","catalog-source-mimic"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"RNA splice-site mapping","scope_note":"Predict splice-site classes from transcript sequence, using a held-out gene split.","historical_missing_metadata":{"protocol_version":"not_yet_extracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"Splice-site mapping evaluates where splice-related labels occur along a sequence. It is distinct from scoring the effect of a particular variant.","summary_source_ids":["src-discovery-terry-r123-rnabenchmark"],"summary_source_locator":"README: Tasks and Datasets (SpliceAI task); model inventory including SpliceBERT variants","sections":[{"title":"Choosing an evaluation","body":"Choose the task’s coordinate system, sequence window and label definition before interpreting scores. BEACON lists a SpliceAI-labelled downstream task and several splice-oriented representations. Their presence in the suite identifies a task area, not a shared checkpoint or proof of performance for every listed model.","source_ids":["src-discovery-terry-r123-rnabenchmark"],"source_locator":"README: Tasks and Datasets (SpliceAI task); model inventory including SpliceBERT variants"},{"title":"Task scope","body":"This is a task guide, not a single versioned benchmark protocol. The connected resources provide examples or concrete procedures. A candidate method or proposed control is not evidence that an evaluation has been completed.","source_ids":["src-discovery-terry-r123-rnabenchmark"],"source_locator":"README: Tasks and Datasets (SpliceAI task); model inventory including SpliceBERT variants"}],"facts":[{"label":"Entity type","value":"Task guide; concrete protocol identities remain separate.","status":"source_checked","source_ids":["src-discovery-terry-r123-rnabenchmark"],"source_locator":"README: Tasks and Datasets (SpliceAI task); model inventory including SpliceBERT variants"},{"label":"Datasets","value":"No single dataset is fixed by this guide. Select a linked protocol and its versioned data release.","status":"inapplicable","source_ids":["src-discovery-terry-r123-rnabenchmark"],"source_locator":"README: Tasks and Datasets (SpliceAI task); model inventory including SpliceBERT variants"},{"label":"Organisms","value":"No shared organism population is defined at this guide level. Record it for each selected dataset.","status":"inapplicable","source_ids":["src-discovery-terry-r123-rnabenchmark"],"source_locator":"README: Tasks and Datasets (SpliceAI task); model inventory including SpliceBERT variants"},{"label":"Assays","value":"No single measurement assay is fixed by this guide; the endpoint and assay belong to the selected protocol.","status":"inapplicable","source_ids":["src-discovery-terry-r123-rnabenchmark"],"source_locator":"README: Tasks and Datasets (SpliceAI task); model inventory including SpliceBERT variants"},{"label":"Splits","value":"No executable split is attached to this task identity. Use the selected protocol’s split manifest.","status":"inapplicable","source_ids":["src-discovery-terry-r123-rnabenchmark"],"source_locator":"README: Tasks and Datasets (SpliceAI task); model inventory including SpliceBERT variants"},{"label":"Allowed inputs","value":"Sequence windows and position-level splice labels under the selected task’s coordinate and strand conventions.","status":"source_checked","source_ids":["src-discovery-terry-r123-rnabenchmark"],"source_locator":"README: Tasks and Datasets (SpliceAI task); model inventory including SpliceBERT variants"},{"label":"Adaptation","value":"No common fitting regime is imposed here. Keep pretrained, frozen, probed, fine-tuned and conventional methods distinct where applicable.","status":"inapplicable","source_ids":["src-discovery-terry-r123-rnabenchmark"],"source_locator":"README: Tasks and Datasets (SpliceAI task); model inventory including SpliceBERT variants"},{"label":"Metrics","value":"The chosen protocol must define site-level scoring, matching tolerance and class balance; this guide does not supply one universal metric.","status":"source_checked","source_ids":["src-discovery-terry-r123-rnabenchmark"],"source_locator":"README: Tasks and Datasets (SpliceAI task); model inventory including SpliceBERT variants"},{"label":"Baselines","value":"Use the protocol’s sequence and splice-specialist comparators with matched inputs. Proposed applicability is separate from completed evaluation.","status":"source_checked","source_ids":["src-discovery-terry-r123-rnabenchmark"],"source_locator":"README: Tasks and Datasets (SpliceAI task); model inventory including SpliceBERT variants"}],"strengths":[{"text":"Position-level labels test localisation, rather than only a sequence-wide classification.","source_ids":["src-discovery-terry-r123-rnabenchmark"],"source_locator":"README: Tasks and Datasets (SpliceAI task); model inventory including SpliceBERT variants"}],"limitations":[{"text":"A splice-site score is not a measured variant-effect score and does not establish clinical interpretation.","source_ids":["src-discovery-terry-r123-rnabenchmark"],"source_locator":"README: Tasks and Datasets (SpliceAI task); model inventory including SpliceBERT variants"}],"diagram":{"title":"Conceptual evaluation workflow","steps":["Pin site labels and coordinate system","Define sequence windows and split","Predict positional splice outputs","Score sites using the declared rule"],"caption":"Conceptual task guide. Dataset preparation, parameters and scoring must come from a separately identified protocol.","source_ids":["src-discovery-terry-r123-rnabenchmark"],"source_locator":"README: Tasks and Datasets (SpliceAI task); model inventory including SpliceBERT variants"},"coverage":"reviewed","gaps":["A concrete evaluation still needs a protocol version, dataset release, eligible/scored counts and documented uncertainty. This guide does not manufacture those run-specific values."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed named primary documentation and existing protocol records. Editorial task scope is distinguished from source-reported procedures. No model execution or new numerical result; a reviewed guide is not a complete runnable protocol."}}}} {"id":"catalog-task-rna-tertiary-structure","kind":"benchmark","name":"RNA tertiary structure","description":"","status":"discovered","facets":{"areas":["rna-transcriptomes"]},"source_ids":["catalog-source-rna-fm","catalog-source-rhofold"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"RNA tertiary structure","scope_note":"Compare predicted 3D folds against independently held-out structures.","historical_missing_metadata":{"protocol_version":"not_yet_extracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"RNA tertiary-structure evaluation compares predicted three-dimensional RNA geometry with reference structures.","summary_source_ids":["src-discovery-ml4bio-rna-fm","evidence-alphafold-paper"],"summary_source_locator":"RNA-FM README: RhoFold; AlphaFold 3 Methods: Nucleic acid prediction baseline and CASP15 RNA targets","sections":[{"title":"Choosing an evaluation","body":"Declare the target molecules and available inputs, including whether an alignment is allowed. Compare coordinates against the relevant reference state using an explicit alignment and atom-selection rule. The AlphaFold 3 paper describes a CASP15 RNA comparison; the RNA-FM project documents RhoFold as a separate complete predictor. A base RNA embedding model is not itself that evaluated pipeline.","source_ids":["src-discovery-ml4bio-rna-fm","evidence-alphafold-paper"],"source_locator":"RNA-FM README: RhoFold; AlphaFold 3 Methods: Nucleic acid prediction baseline and CASP15 RNA targets"},{"title":"Task scope","body":"This is a task guide, not a single versioned benchmark protocol. The connected resources provide examples or concrete procedures. A candidate method or proposed control is not evidence that an evaluation has been completed.","source_ids":["src-discovery-ml4bio-rna-fm","evidence-alphafold-paper"],"source_locator":"RNA-FM README: RhoFold; AlphaFold 3 Methods: Nucleic acid prediction baseline and CASP15 RNA targets"}],"facts":[{"label":"Entity type","value":"Task guide; concrete protocol identities remain separate.","status":"source_checked","source_ids":["src-discovery-ml4bio-rna-fm","evidence-alphafold-paper"],"source_locator":"RNA-FM README: RhoFold; AlphaFold 3 Methods: Nucleic acid prediction baseline and CASP15 RNA targets"},{"label":"Datasets","value":"No single dataset is fixed by this guide. Select a linked protocol and its versioned data release.","status":"inapplicable","source_ids":["src-discovery-ml4bio-rna-fm","evidence-alphafold-paper"],"source_locator":"RNA-FM README: RhoFold; AlphaFold 3 Methods: Nucleic acid prediction baseline and CASP15 RNA targets"},{"label":"Organisms","value":"No shared organism population is defined at this guide level. Record it for each selected dataset.","status":"inapplicable","source_ids":["src-discovery-ml4bio-rna-fm","evidence-alphafold-paper"],"source_locator":"RNA-FM README: RhoFold; AlphaFold 3 Methods: Nucleic acid prediction baseline and CASP15 RNA targets"},{"label":"Assays","value":"No single measurement assay is fixed by this guide; the endpoint and assay belong to the selected protocol.","status":"inapplicable","source_ids":["src-discovery-ml4bio-rna-fm","evidence-alphafold-paper"],"source_locator":"RNA-FM README: RhoFold; AlphaFold 3 Methods: Nucleic acid prediction baseline and CASP15 RNA targets"},{"label":"Splits","value":"No executable split is attached to this task identity. Use the selected protocol’s split manifest.","status":"inapplicable","source_ids":["src-discovery-ml4bio-rna-fm","evidence-alphafold-paper"],"source_locator":"RNA-FM README: RhoFold; AlphaFold 3 Methods: Nucleic acid prediction baseline and CASP15 RNA targets"},{"label":"Allowed inputs","value":"RNA sequence and permitted auxiliary information, plus a reference structure for geometric evaluation.","status":"source_checked","source_ids":["src-discovery-ml4bio-rna-fm","evidence-alphafold-paper"],"source_locator":"RNA-FM README: RhoFold; AlphaFold 3 Methods: Nucleic acid prediction baseline and CASP15 RNA targets"},{"label":"Adaptation","value":"No common fitting regime is imposed here. Keep pretrained, frozen, probed, fine-tuned and conventional methods distinct where applicable.","status":"inapplicable","source_ids":["src-discovery-ml4bio-rna-fm","evidence-alphafold-paper"],"source_locator":"RNA-FM README: RhoFold; AlphaFold 3 Methods: Nucleic acid prediction baseline and CASP15 RNA targets"},{"label":"Metrics","value":"Protocol-defined structural accuracy. Atom selection, reference states and treatment of missing residues must accompany a number.","status":"source_checked","source_ids":["src-discovery-ml4bio-rna-fm","evidence-alphafold-paper"],"source_locator":"RNA-FM README: RhoFold; AlphaFold 3 Methods: Nucleic acid prediction baseline and CASP15 RNA targets"},{"label":"Baselines","value":"Use complete RNA-structure predictors and their stated input information; a language-model family name alone does not identify a structural pipeline.","status":"source_checked","source_ids":["src-discovery-ml4bio-rna-fm","evidence-alphafold-paper"],"source_locator":"RNA-FM README: RhoFold; AlphaFold 3 Methods: Nucleic acid prediction baseline and CASP15 RNA targets"}],"strengths":[{"text":"Coordinate-level scoring tests a structural endpoint beyond nucleotide pairing or embedding quality.","source_ids":["src-discovery-ml4bio-rna-fm","evidence-alphafold-paper"],"source_locator":"RNA-FM README: RhoFold; AlphaFold 3 Methods: Nucleic acid prediction baseline and CASP15 RNA targets"}],"limitations":[{"text":"Alternative conformations and incomplete experimental structures complicate reference comparison. Predictions are not a calibrated dynamic ensemble.","source_ids":["src-discovery-ml4bio-rna-fm","evidence-alphafold-paper"],"source_locator":"RNA-FM README: RhoFold; AlphaFold 3 Methods: Nucleic acid prediction baseline and CASP15 RNA targets"}],"diagram":{"title":"Conceptual evaluation workflow","steps":["Select targets and reference states","Declare inputs and full prediction pipeline","Predict RNA coordinates","Align and score the specified atoms"],"caption":"Conceptual task guide. Dataset preparation, parameters and scoring must come from a separately identified protocol.","source_ids":["src-discovery-ml4bio-rna-fm","evidence-alphafold-paper"],"source_locator":"RNA-FM README: RhoFold; AlphaFold 3 Methods: Nucleic acid prediction baseline and CASP15 RNA targets"},"coverage":"reviewed","gaps":["A concrete evaluation still needs a protocol version, dataset release, eligible/scored counts and documented uncertainty. This guide does not manufacture those run-specific values."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed named primary documentation and existing protocol records. Editorial task scope is distinguished from source-reported procedures. No model execution or new numerical result; a reviewed guide is not a complete runnable protocol."}}}} {"id":"catalog-task-utr-translation","kind":"benchmark","name":"Translation / RNA stability","description":"","status":"discovered","facets":{"areas":["rna-transcriptomes"]},"source_ids":["catalog-source-mrna-fm","catalog-source-mimic"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"Translation / RNA stability","scope_note":"Predict measured translation or stability effects; choose UTR or coding-sequence assays to match each model’s input modality.","historical_missing_metadata":{"protocol_version":"not_yet_extracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"Translation and RNA-stability tasks test different measurable properties of transcripts and should be reported as separate endpoints.","summary_source_ids":["src-discovery-morrislab-mrnabench"],"summary_source_locator":"README: Dataset Catalog / Translation Regulation and RNA Stability; Model Catalog / Baseline Models","sections":[{"title":"Choosing an evaluation","body":"Select the endpoint and biological context before comparing methods. mRNABench distinguishes mean ribosome load, translation efficiency and RNA half-life datasets. Reporter constructs, native transcripts and different organisms are not interchangeable datasets. Record which transcript regions and supervision are available to the model.","source_ids":["src-discovery-morrislab-mrnabench"],"source_locator":"README: Dataset Catalog / Translation Regulation and RNA Stability; Model Catalog / Baseline Models"},{"title":"Task scope","body":"This is a task guide, not a single versioned benchmark protocol. The connected resources provide examples or concrete procedures. A candidate method or proposed control is not evidence that an evaluation has been completed.","source_ids":["src-discovery-morrislab-mrnabench"],"source_locator":"README: Dataset Catalog / Translation Regulation and RNA Stability; Model Catalog / Baseline Models"}],"facts":[{"label":"Entity type","value":"Task guide; concrete protocol identities remain separate.","status":"source_checked","source_ids":["src-discovery-morrislab-mrnabench"],"source_locator":"README: Dataset Catalog / Translation Regulation and RNA Stability; Model Catalog / Baseline Models"},{"label":"Datasets","value":"No single dataset is fixed by this guide. Select a linked protocol and its versioned data release.","status":"inapplicable","source_ids":["src-discovery-morrislab-mrnabench"],"source_locator":"README: Dataset Catalog / Translation Regulation and RNA Stability; Model Catalog / Baseline Models"},{"label":"Organisms","value":"No shared organism population is defined at this guide level. Record it for each selected dataset.","status":"inapplicable","source_ids":["src-discovery-morrislab-mrnabench"],"source_locator":"README: Dataset Catalog / Translation Regulation and RNA Stability; Model Catalog / Baseline Models"},{"label":"Assays","value":"No single measurement assay is fixed by this guide; the endpoint and assay belong to the selected protocol.","status":"inapplicable","source_ids":["src-discovery-morrislab-mrnabench"],"source_locator":"README: Dataset Catalog / Translation Regulation and RNA Stability; Model Catalog / Baseline Models"},{"label":"Splits","value":"No executable split is attached to this task identity. Use the selected protocol’s split manifest.","status":"inapplicable","source_ids":["src-discovery-morrislab-mrnabench"],"source_locator":"README: Dataset Catalog / Translation Regulation and RNA Stability; Model Catalog / Baseline Models"},{"label":"Allowed inputs","value":"Transcript or UTR sequences with task-specific experimental measurements and declared transcript-region boundaries.","status":"source_checked","source_ids":["src-discovery-morrislab-mrnabench"],"source_locator":"README: Dataset Catalog / Translation Regulation and RNA Stability; Model Catalog / Baseline Models"},{"label":"Adaptation","value":"No common fitting regime is imposed here. Keep pretrained, frozen, probed, fine-tuned and conventional methods distinct where applicable.","status":"inapplicable","source_ids":["src-discovery-morrislab-mrnabench"],"source_locator":"README: Dataset Catalog / Translation Regulation and RNA Stability; Model Catalog / Baseline Models"},{"label":"Metrics","value":"Endpoint-specific regression or classification as defined by the selected dataset; retain measurement units and any target transformation.","status":"source_checked","source_ids":["src-discovery-morrislab-mrnabench"],"source_locator":"README: Dataset Catalog / Translation Regulation and RNA Stability; Model Catalog / Baseline Models"},{"label":"Baselines","value":"mRNABench supplies conventional baselines alongside RNA and DNA representation models. Compare only within the same endpoint, split and input scope.","status":"source_checked","source_ids":["src-discovery-morrislab-mrnabench"],"source_locator":"README: Dataset Catalog / Translation Regulation and RNA Stability; Model Catalog / Baseline Models"}],"strengths":[{"text":"Separate molecular endpoints allow evaluation of specific transcript properties instead of an undefined general RNA score.","source_ids":["src-discovery-morrislab-mrnabench"],"source_locator":"README: Dataset Catalog / Translation Regulation and RNA Stability; Model Catalog / Baseline Models"}],"limitations":[{"text":"Ribosome loading, translation efficiency and stability are related but distinct measurements. Good performance on one is not evidence for the others.","source_ids":["src-discovery-morrislab-mrnabench"],"source_locator":"README: Dataset Catalog / Translation Regulation and RNA Stability; Model Catalog / Baseline Models"}],"diagram":{"title":"Conceptual evaluation workflow","steps":["Choose molecular endpoint and dataset","Fix sequence regions and held-out split","Fit or apply the declared predictor","Score the matching measurement"],"caption":"Conceptual task guide. Dataset preparation, parameters and scoring must come from a separately identified protocol.","source_ids":["src-discovery-morrislab-mrnabench"],"source_locator":"README: Dataset Catalog / Translation Regulation and RNA Stability; Model Catalog / Baseline Models"},"coverage":"reviewed","gaps":["A concrete evaluation still needs a protocol version, dataset release, eligible/scored counts and documented uncertainty. This guide does not manufacture those run-specific values."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed named primary documentation and existing protocol records. Editorial task scope is distinguished from source-reported procedures. No model execution or new numerical result; a reviewed guide is not a complete runnable protocol."}}}} {"id":"cathe2-2025","kind":"source","name":"CATHe2: Enhanced CATH superfamily detection using ProstT5 and structural alphabets","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["proteins-complexes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12631783/","version":"journal full text in PMC","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.1093/biomethods/bpaf080","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"713dbfb6ec1cc1aa85c0543eb93aafa0b45b8873df28053b765dd0a1b6d9b563","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12631783/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:38.366Z","legacy_paper":{"id":"cathe2-2025","title":"CATHe2: Enhanced CATH superfamily detection using ProstT5 and structural alphabets","year":2025,"publication_status":"peer_reviewed","version":"journal full text in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12631783/","primary_domain":"proteins-complexes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.1093/biomethods/bpaf080","notes":"Numeric result checked against Table 3. in primary full-text XML; journal/source: Biology Methods & Protocols."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"cell-dino-2025","kind":"source","name":"Cell-DINO: Self-supervised image-based embeddings for cell fluorescent microscopy","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["cells-tissues"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12826486/","version":"version of record","retrieved_at":"2026-09-15T23:37:05Z","doi":"10.1371/journal.pcbi.1013828","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"12a53a78c70b3033c3351cf7afd4da42ebc98bb3281308f07e71e5baffc153a0","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12826486/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:06Z","legacy_paper":{"id":"cell-dino-2025","title":"Cell-DINO: Self-supervised image-based embeddings for cell fluorescent microscopy","year":2025,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12826486/","primary_domain":"cells-tissues","retrieved_utc":"2026-09-15T23:37:05Z","notes":"Primary full text verified via Europe PMC fullTextXML; venue: PLOS Computational Biology; PMC ID: PMC12826486. PL column is F1-score reported on a 0–100 scale; Cell-DINO is a vision encoder plus downstream classifier.","doi":"10.1371/journal.pcbi.1013828"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"cell2sentence-2024","kind":"source","name":"Cell2Sentence: Teaching Large Language Models the Language of Biology","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["cells-tissues"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11565894/","version":"preprint archived 2024-10-29","retrieved_at":"2026-09-15T23:25:00Z","doi":"10.1101/2023.09.11.557287","publication_status":"preprint","year":2024,"artifact_sha256":"e088727d6e04857fccb7033a9b074e1850f775e86e7d2e99e603dde09558ab02","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11565894/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:16.533640+00:00","legacy_paper":{"id":"cell2sentence-2024","title":"Cell2Sentence: Teaching Large Language Models the Language of Biology","year":2024,"publication_status":"preprint","version":"preprint archived 2024-10-29","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11565894/","primary_domain":"cells-tissues","retrieved_utc":"2026-09-15T23:25:00Z","notes":"Primary full text via Europe PMC XML; venue: bioRxiv; PMC ID: PMC11565894.","doi":"10.1101/2023.09.11.557287"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"claim-b2-2ome-lm-2025","kind":"claim","name":"Reported AUC for 2OMe-LM","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["human RNA 2-prime-O-methylation site prediction"]},"source_ids":["2ome-lm-2025"],"links":[{"relation":"subject","target_id":"b2-2ome-lm-2025"}],"attributes":{"field":"attributes.printed_value","value":"0.919","source_locator":"Table 1, 2OMe-LM row, AUC column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:38.332Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-b2-antibody-deamidation-plm-2024","kind":"claim","name":"Reported accuracy for ESM-2 650M embeddings + classifier","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["antibody deamidation-site prediction"]},"source_ids":["antibody-deamidation-plm-2024"],"links":[{"relation":"subject","target_id":"b2-antibody-deamidation-plm-2024"}],"attributes":{"field":"attributes.printed_value","value":"0.944","source_locator":"Table 1, Global embeddings only row, Accuracy column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:38.478Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-b2-barcodebert-2026","kind":"claim","name":"Reported accuracy for BarcodeBERT (4–4-4)","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["unseen-species genus classification"]},"source_ids":["barcodebert-2026"],"links":[{"relation":"subject","target_id":"b2-barcodebert-2026"}],"attributes":{"field":"attributes.printed_value","value":"78.5","source_locator":"Table 1, BarcodeBERT (4–4-4) row, unseen-species genus-level 1-NN Acc (%) column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558051+00:00","notes":"Resolved the two-level column header: Acc (%) falls under genus-level 1-NN probe of unseen species, not seen-species classification or BIN reconstruction. BarcodeBERT (4–4-4) has 78.5 in this cell."}}} {"id":"claim-b2-birna-bert-2025","kind":"claim","name":"Reported F1 for BiRNA-BERT","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["extremely long RNA species classification"]},"source_ids":["birna-bert-2025"],"links":[{"relation":"subject","target_id":"b2-birna-bert-2025"}],"attributes":{"field":"attributes.printed_value","value":"0.804","source_locator":"Table 2, BiRNA-BERT row, F1 Score column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:38.292Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-b2-cathe2-2025","kind":"claim","name":"Reported F1 for CATHe2 + ProstT5","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["CATH superfamily annotation"]},"source_ids":["cathe2-2025"],"links":[{"relation":"subject","target_id":"b2-cathe2-2025"}],"attributes":{"field":"attributes.printed_value","value":"82.3","source_locator":"Table 3, ProstT5 full row, F1 score column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:38.366Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-b2-clathrin-plm-2025","kind":"claim","name":"Reported accuracy for ESM-2 embedding + paper classifier","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["clathrin protein classification"]},"source_ids":["clathrin-plm-2025"],"links":[{"relation":"subject","target_id":"b2-clathrin-plm-2025"}],"attributes":{"field":"attributes.printed_value","value":"0.916","source_locator":"Table 2, Independent test / ESM-2 row, ACC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558194+00:00","notes":"Resolved the blank evaluation-strategy cells by their independent-test row group. ESM-2 ACC is 0.916 there; the cross-validation ESM-2 ACC is instead 0.873. This is the paper classifier using embeddings, not a standalone checkpoint."}}} {"id":"claim-b2-cobra-rna-binding-2026","kind":"claim","name":"Reported MCC for ERNIE-RNA + CoBRA","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA compound-binding site prediction"]},"source_ids":["cobra-rna-binding-2026"],"links":[{"relation":"subject","target_id":"b2-cobra-rna-binding-2026"}],"attributes":{"field":"attributes.printed_value","value":"0.657","source_locator":"Table 2, ERNIE-RNA / TCL focal row, MCC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558197+00:00","notes":"Matched ERNIE-RNA jointly with TCL focal loss, then the MCC column. Table 2 explicitly reports test-set models. The cell is 0.657, distinct from AUROC 0.868."}}} {"id":"claim-b2-codonbert-vaccines-2024","kind":"claim","name":"Reported Spearman rho for CodonBERT","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["flu-vaccine mRNA property prediction"]},"source_ids":["codonbert-vaccines-2024"],"links":[{"relation":"subject","target_id":"b2-codonbert-vaccines-2024"}],"attributes":{"field":"attributes.printed_value","value":"0.81","source_locator":"Table 2, CodonBERT row, Flu vaccines Spearman correlation column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558201+00:00","notes":"Matched the CodonBERT row and Flu vaccines column (0.81). The table footnote identifies regression columns as Spearman rank correlation and singles out E. coli as classification; this is not a flu-vaccine accuracy score."}}} {"id":"claim-b2-dart-eval-regulatory-2024","kind":"claim","name":"Reported accuracy for DNABERT-2","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["regulatory element identification"]},"source_ids":["dart-eval-regulatory-2024"],"links":[{"relation":"subject","target_id":"b2-dart-eval-regulatory-2024"}],"attributes":{"field":"attributes.printed_value","value":"0.876","source_locator":"Table 3 (PDF page 5), DNABERT-2 row, Zero-Shot Accuracy column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558203+00:00","notes":"Inspected the pinned NeurIPS primary PDF table and explanatory text. The DNABERT-2 row reports 0.876 under Zero-Shot Accuracy. The caption defines this as pairwise prioritization of positives over matched controls, distinct from supervised absolute accuracy."}}} {"id":"claim-b2-dnabert2-enhancer-2025","kind":"claim","name":"Reported AUC for DNABERT2-Enhancer","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["enhancer recognition"]},"source_ids":["dnabert2-enhancer-2025"],"links":[{"relation":"subject","target_id":"b2-dnabert2-enhancer-2025"}],"attributes":{"field":"attributes.printed_value","value":"0.965","source_locator":"Table 4, first-layer DNABERT2-Enhancer row, AUC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558204+00:00","notes":"Resolved the first-layer row group. DNABERT2-Enhancer AUC is 0.965, whereas second-layer AUC is 0.933. The caption explicitly describes 5-fold cross-validation on Liu training data, not an independent held-out test."}}} {"id":"claim-b2-eden-genomic-classification-2026","kind":"claim","name":"Reported MCC for DNABERT-2","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["human core-promoter classification"]},"source_ids":["eden-genomic-classification-2026"],"links":[{"relation":"subject","target_id":"b2-eden-genomic-classification-2026"}],"attributes":{"field":"attributes.printed_value","value":"70.52","source_locator":"Table 5, DNABERT-2 row, H-CPD (MCC) column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:37.531Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-b2-ernie-rna-2025","kind":"claim","name":"Reported binary F1 for ERNIE-RNA","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary-structure prediction"]},"source_ids":["ernie-rna-2025"],"links":[{"relation":"subject","target_id":"b2-ernie-rna-2025"}],"attributes":{"field":"attributes.printed_value","value":"0.575","source_locator":"Table 2, ERNIE-RNA zero-shot row, bpRNA-new F1-Score (binary) column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558206+00:00","notes":"Resolved bpRNA-new as the first three-column dataset group and F1-Score (binary) as its third metric. ERNIE-RNA zero shot is 86M and reports 0.575; RNA3DB-2D F1 is instead 0.542."}}} {"id":"claim-b2-esm2-ofs-fitness-2025","kind":"claim","name":"Reported Spearman rho for ESM2 OFS pseudo-perplexity","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["protein variant fitness prediction"]},"source_ids":["esm2-ofs-fitness-2025"],"links":[{"relation":"subject","target_id":"b2-esm2-ofs-fitness-2025"}],"attributes":{"field":"attributes.printed_value","value":"0.403","source_locator":"Published PDF page 6 (033014-6), Table I, ESM2: OFS PP row, Aggregate mean column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source review, not human review","reviewed_at":"2026-09-16T10:45:41.099916+00:00","notes":"Publisher PDF retrieved through official APS harvest endpoint after direct download returned403. Table I is ProteinGym substitutions, not indels TableII. Last column aggregate mean0.403; separate function categories precede it. This verifies reported score, not experimental reproduction. Comparator rows in this table are sourced from ProteinGym; OFS PP is authors own method."}}} {"id":"claim-b2-fusion-breakpoint-foundation-models-2026","kind":"claim","name":"Reported ROC AUC for Nucleotide Transformer + NN (middle)","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["gene fusion breakpoint classification"]},"source_ids":["fusion-breakpoint-foundation-models-2026"],"links":[{"relation":"subject","target_id":"b2-fusion-breakpoint-foundation-models-2026"}],"attributes":{"field":"attributes.printed_value","value":"0.994","source_locator":"Table 2, NT / NN (middle) row, ROC AUC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558209+00:00","notes":"Matched NT jointly with NN (middle) and ROC AUC 0.994 in the full-test-set table. NT with SVM reports 0.995 and is a separate pipeline."}}} {"id":"claim-b2-genomic-tokenizer-selection-2025","kind":"claim","name":"Reported MCC for Caduceus (character tokens)","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["regulatory sequence classification"]},"source_ids":["genomic-tokenizer-selection-2025"],"links":[{"relation":"subject","target_id":"b2-genomic-tokenizer-selection-2025"}],"attributes":{"field":"attributes.printed_value","value":"0.778","source_locator":"Table 2, Regulatory row, Caduceus (char) MCC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558210+00:00","notes":"Matched Regulatory row with Caduceus (char) column, 0.778. Caption establishes these as MCC summaries by category; model-size row identifies 3.9M parameters. This is an aggregated category result, not a single unspecified split."}}} {"id":"claim-b2-gsmformer-ppi-2026","kind":"claim","name":"Reported AUROC for GSMFormer-PPI + ProstT5","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["protein-protein interaction prediction"]},"source_ids":["gsmformer-ppi-2026"],"links":[{"relation":"subject","target_id":"b2-gsmformer-ppi-2026"}],"attributes":{"field":"attributes.printed_value","value":"0.988","source_locator":"Table 6, ProstT5 embedding row, AUROC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558212+00:00","notes":"Matched ProstT5 embedding row and AUROC column, 0.988. Caption explicitly describes GSMFormer-PPI using embeddings as node features, not standalone ProstT5 prediction."}}} {"id":"claim-b2-megsite-2025","kind":"claim","name":"Reported AUC for MegSite + ESM3","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["DNA-binding residue prediction"]},"source_ids":["megsite-2025"],"links":[{"relation":"subject","target_id":"b2-megsite-2025"}],"attributes":{"field":"attributes.printed_value","value":"0.948","source_locator":"Table 2, DNA-129_Test / ESM3 row, AUC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558213+00:00","notes":"Resolved DNA-129_Test row group and ESM3 row. AUC is 0.948; the next numeric cell 0.582 is AP. Caption states an embedding comparison within MegSite."}}} {"id":"claim-b2-mrna-lm-2025","kind":"claim","name":"Reported Spearman rho for mRNA-LM","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["mRNA half-life prediction"]},"source_ids":["mrna-lm-2025"],"links":[{"relation":"subject","target_id":"b2-mrna-lm-2025"}],"attributes":{"field":"attributes.printed_value","value":"0.696","source_locator":"Table 1, mRNA-LM row, mRNA half-life Spearman column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558214+00:00","notes":"Resolved mRNA half-life column under the Spearman header spanning three tasks. mRNA-LM gives 0.696. Caption identifies average test performance across cross-validation splits; protein-expression AUROC is a different column."}}} {"id":"claim-b2-mrnabert-2025","kind":"claim","name":"Reported R-squared for mRNABERT","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["translation-efficiency prediction"]},"source_ids":["mrnabert-2025"],"links":[{"relation":"subject","target_id":"b2-mrnabert-2025"}],"attributes":{"field":"attributes.printed_value","value":"0.669","source_locator":"Table 2, mRNABERT (3066) row, Human R-squared column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558216+00:00","notes":"Resolved Human group and its R-squared subcolumn. mRNABERT (3066) reports 0.669; Human Spearman is 0.814 and Mouse R-squared is 0.649. Caption specifies ultra-long mRNA translation-efficiency prediction."}}} {"id":"claim-b2-mulan-2025","kind":"claim","name":"Reported AUC for MULAN-ESM2 S","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["human protein-protein interaction prediction"]},"source_ids":["mulan-2025"],"links":[{"relation":"subject","target_id":"b2-mulan-2025"}],"attributes":{"field":"attributes.printed_value","value":"0.717","source_locator":"Table 2, MULAN-ESM2 S row, HumanPPI AUC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558217+00:00","notes":"Resolved the multirow header: HumanPPI uses AUC. MULAN-ESM2 S has 0.717; this is the small-model group, distinct from M and L variants."}}} {"id":"claim-b2-phylogpn-2025","kind":"claim","name":"Reported AUROC for PhyloGPN","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["ClinVar 3-prime UTR variant classification"]},"source_ids":["phylogpn-2025"],"links":[{"relation":"subject","target_id":"b2-phylogpn-2025"}],"attributes":{"field":"attributes.printed_value","value":"0.94","source_locator":"Table 1, 3-prime UTR row, PhyloGPN AUROC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558218+00:00","notes":"Matched 3-prime UTR row and PhyloGPN column (0.94). Caption specifies log-likelihood-ratio predictions of ClinVar classes and explicitly defines each cell as AUROC."}}} {"id":"claim-b2-polya-glm-2025","kind":"claim","name":"Reported AUC for HyenaDNA","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["polyadenylation site detection"]},"source_ids":["polya-glm-2025"],"links":[{"relation":"subject","target_id":"b2-polya-glm-2025"}],"attributes":{"field":"attributes.printed_value","value":"0.7510","source_locator":"Table 1, Few-shot HyenaDNA row, G-G AUC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558220+00:00","notes":"Resolved Few-shot group, HyenaDNA row, and G-G subcolumn under AUC (0.7510). IG-G AUC is 0.7541. Caption states averages over five-fold cross-validation and distinguishes negative sampling regions."}}} {"id":"claim-b2-rlsite-rna-binding-2025","kind":"claim","name":"Reported AUC for RLsite","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA-small-molecule binding-site prediction"]},"source_ids":["rlsite-rna-binding-2025"],"links":[{"relation":"subject","target_id":"b2-rlsite-rna-binding-2025"}],"attributes":{"field":"attributes.printed_value","value":"0.828","source_locator":"Table 1, RLsite row, T18 AUC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558222+00:00","notes":"Matched RLsite and AUC (0.828). Caption explicitly identifies dataset T18; MCC 0.474 is a different metric."}}} {"id":"claim-b2-rnaret-2026","kind":"claim","name":"Reported F1 for RNAret","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["miRNA-mRNA interaction prediction"]},"source_ids":["rnaret-2026"],"links":[{"relation":"subject","target_id":"b2-rnaret-2026"}],"attributes":{"field":"attributes.printed_value","value":"0.9622","source_locator":"Table 1, MirTarRAW / 5-mer RNAret row, F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558224+00:00","notes":"Resolved the MirTarRAW section, 5-mer RNAret row, and F1 column (0.9622), distinct from DeepMirTarLeft F1 0.9728. Methods confirm 72/8/20 train/validation/test partition for MirTarRAW."}}} {"id":"claim-b2-spin-protein-function-2026","kind":"claim","name":"Reported F1 macro-weighted for SPIN + ESM2-35M","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["protein function annotation"]},"source_ids":["spin-protein-function-2026"],"links":[{"relation":"subject","target_id":"b2-spin-protein-function-2026"}],"attributes":{"field":"attributes.printed_value","value":"0.796","source_locator":"Table 1, ESM2-35M Test row, F1_m-w column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558225+00:00","notes":"Resolved Test group and macro-weighted F1 subcolumn (0.796) for frozen ESM2-35M in SPIN. Test weighted accuracy is 0.798. Methods define inverse-frequency class weighting for macro-weighted F1."}}} {"id":"claim-b2-structure-informed-plm-2025","kind":"claim","name":"Reported AUROC for structure-informed pLM","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["protein variant-effect classification"]},"source_ids":["structure-informed-plm-2025"],"links":[{"relation":"subject","target_id":"b2-structure-informed-plm-2025"}],"attributes":{"field":"attributes.printed_value","value":".803","source_locator":"PMC12068927 HTML, Table4, AA+SS+RSA+CM row, AUROC column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source review, not human review","reviewed_at":"2026-09-16T10:45:41.099916+00:00","notes":"Full-text HTML succeeds although EuropePMC XMLreturned404. Row is mutation-site variables AA+SS+RSA+CM, not neighbouring environment variant. AUROC .803 is numerically equivalent to preserved legacy0.803. Source check, not experimental reproduction; do not claim original source printed leading zero."}}} {"id":"claim-lit-001","kind":"claim","name":"Reported AUC for Caduceus-Ph","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["Human 5mC detection"]},"source_ids":["dna-foundation-models-2025"],"links":[{"relation":"subject","target_id":"lit-001"}],"attributes":{"field":"attributes.printed_value","value":"0.783","source_locator":"Table 3, Human 5mC row, Caduceus-Ph column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:35.327Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-002","kind":"claim","name":"Reported AUC for NT-v2","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["Human 5mC detection"]},"source_ids":["dna-foundation-models-2025"],"links":[{"relation":"subject","target_id":"lit-002"}],"attributes":{"field":"attributes.printed_value","value":"0.7377","source_locator":"Table 3, Human 5mC row, NT-v2 column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:35.327Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-003","kind":"claim","name":"Reported Accuracy for ENBED","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["Enhancer classification"]},"source_ids":["enbed-2024"],"links":[{"relation":"subject","target_id":"lit-003"}],"attributes":{"field":"attributes.printed_value","value":"90.3","source_locator":"Table 2, Mouse Enhancers row, ENBED column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:35.378Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-004","kind":"claim","name":"Reported Accuracy for ENBED (GRCh38)","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["Enhancer classification"]},"source_ids":["enbed-2024"],"links":[{"relation":"subject","target_id":"lit-004"}],"attributes":{"field":"attributes.printed_value","value":"81.1","source_locator":"Table 2, Mouse Enhancers row, ENBED (GRCh38) column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:35.378Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-005","kind":"claim","name":"Reported Accuracy for DNABERT-2","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["G-quadruplex classification"]},"source_ids":["quadruplex-llm-benchmark-2025"],"links":[{"relation":"subject","target_id":"lit-005"}],"attributes":{"field":"attributes.printed_value","value":"97.0","source_locator":"Table 5, DNABERT-2 (117 M) row, Accuracy column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:35.379Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-006","kind":"claim","name":"Reported Accuracy for Caduceus","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["G-quadruplex classification"]},"source_ids":["quadruplex-llm-benchmark-2025"],"links":[{"relation":"subject","target_id":"lit-006"}],"attributes":{"field":"attributes.printed_value","value":"95.0","source_locator":"Table 5, Caduceus (8 M) row, Accuracy column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:35.379Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-007","kind":"claim","name":"Reported AUROC for HyenaDNA","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["Enhancer-target gene prediction"]},"source_ids":["dnalongbench-2025"],"links":[{"relation":"subject","target_id":"lit-007"}],"attributes":{"field":"attributes.printed_value","value":"0.828","source_locator":"Table 3, HyenaDNA row, ETGP column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.492545+00:00","notes":"ETGP is the first numeric column, separate from the six CMP cell-type columns and average; caption defines ETGP AUROC. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-008","kind":"claim","name":"Reported AUROC for Caduceus-Ph","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["Enhancer-target gene prediction"]},"source_ids":["dnalongbench-2025"],"links":[{"relation":"subject","target_id":"lit-008"}],"attributes":{"field":"attributes.printed_value","value":"0.826","source_locator":"Table 3, Caduceus-Ph row, ETGP column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.493765+00:00","notes":"ETGP is the first numeric column, separate from the six CMP cell-type columns and average; caption defines ETGP AUROC. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-009","kind":"claim","name":"Reported Pearson R for RiNALMo","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["Mean ribosome load from MPRA"]},"source_ids":["mrnabench-2025"],"links":[{"relation":"subject","target_id":"lit-009"}],"attributes":{"field":"attributes.printed_value","value":"0.74","source_locator":"Table 2, RiNALMo row, MRL MPRA column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.497221+00:00","notes":"MRL MPRA is the second task under Local and uses R. Caption specifies mean over ten random seeds and selected best model per family, not a fully identified checkpoint. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-010","kind":"claim","name":"Reported Pearson R for RNA-FM","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["Mean ribosome load from MPRA"]},"source_ids":["mrnabench-2025"],"links":[{"relation":"subject","target_id":"lit-010"}],"attributes":{"field":"attributes.printed_value","value":"0.49","source_locator":"Table 2, RNA-FM row, MRL MPRA column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.500211+00:00","notes":"MRL MPRA is the second task under Local and uses R. Caption specifies mean over ten random seeds and selected best model per family, not a fully identified checkpoint. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-011","kind":"claim","name":"Reported F1 for BPfold","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary structure"]},"source_ids":["bpfold-2025"],"links":[{"relation":"subject","target_id":"lit-011"}],"attributes":{"field":"attributes.printed_value","value":"0.814","source_locator":"Table 2, BPfold row, PDB F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.502000+00:00","notes":"PDB is the second four-metric block; its F1 is numeric column six, not Rfam F1. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-012","kind":"claim","name":"Reported F1 for RNAfold","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary structure"]},"source_ids":["bpfold-2025"],"links":[{"relation":"subject","target_id":"lit-012"}],"attributes":{"field":"attributes.printed_value","value":"0.747","source_locator":"Table 2, RNAfold row, PDB F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.504220+00:00","notes":"PDB is the second four-metric block; its F1 is numeric column six, not Rfam F1. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-013","kind":"claim","name":"Reported F1 for TU-Fold (aug)","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary structure"]},"source_ids":["tu-fold-2025"],"links":[{"relation":"subject","target_id":"lit-013"}],"attributes":{"field":"attributes.printed_value","value":"0.947","source_locator":"Table 2, TU-Fold (aug) row, Overall F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.505799+00:00","notes":"Overall is the first two-metric block. F1 is first numeric column; source cell includes uncertainty after the preserved central value. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-014","kind":"claim","name":"Reported F1 for UFold","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary structure"]},"source_ids":["tu-fold-2025"],"links":[{"relation":"subject","target_id":"lit-014"}],"attributes":{"field":"attributes.printed_value","value":"0.938","source_locator":"Table 2, UFold row, Overall F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.507031+00:00","notes":"Overall is the first two-metric block. F1 is first numeric column; source cell includes uncertainty after the preserved central value. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-015","kind":"claim","name":"Reported Median F1 for DEBFold","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary structure"]},"source_ids":["debfold-2024"],"links":[{"relation":"subject","target_id":"lit-015"}],"attributes":{"field":"attributes.printed_value","value":"55.7","source_locator":"Table 1, DEBFold row, TestSetβ F1 (%) column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.509290+00:00","notes":"TestSet beta is the second four-column block. F1 (%) is its first column; caption reports test-set median F1. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-016","kind":"claim","name":"Reported Median F1 for RNAfold","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary structure"]},"source_ids":["debfold-2024"],"links":[{"relation":"subject","target_id":"lit-016"}],"attributes":{"field":"attributes.printed_value","value":"52.3","source_locator":"Table 1, RNAfold row, TestSetβ F1 (%) column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.511166+00:00","notes":"TestSet beta is the second four-column block. F1 (%) is its first column; caption reports test-set median F1. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-017","kind":"claim","name":"Reported Mean Spearman rho for ESM-2","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["Zero-shot substitution mutation effects: stability"]},"source_ids":["proteingym-2023"],"links":[{"relation":"subject","target_id":"lit-017"}],"attributes":{"field":"attributes.printed_value","value":"0.488","source_locator":"Table A7, ESM-2 (15B) row, Stability column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.517323+00:00","notes":"Table A7 is zero-shot substitution DMS grouped by function. Stability is the fifth numeric column; model-type row spans do not change its placement. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-018","kind":"claim","name":"Reported Mean Spearman rho for ProteinMPNN","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["Zero-shot substitution mutation effects: stability"]},"source_ids":["proteingym-2023"],"links":[{"relation":"subject","target_id":"lit-018"}],"attributes":{"field":"attributes.printed_value","value":"0.566","source_locator":"Table A7, ProteinMPNN row, Stability column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.523422+00:00","notes":"Table A7 is zero-shot substitution DMS grouped by function. Stability is the fifth numeric column; model-type row spans do not change its placement. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-019","kind":"claim","name":"Reported AUROC for FUJISAN","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["Enzyme functional identity prediction"]},"source_ids":["fujisan-2024"],"links":[{"relation":"subject","target_id":"lit-019"}],"attributes":{"field":"attributes.printed_value","value":"0.9427","source_locator":"Table 1, FUJISAN row, AUROC column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:35.728Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-020","kind":"claim","name":"Reported AUROC for ESM2","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["Enzyme functional identity prediction"]},"source_ids":["fujisan-2024"],"links":[{"relation":"subject","target_id":"lit-020"}],"attributes":{"field":"attributes.printed_value","value":"0.7991","source_locator":"Table 1, ESM2 row, AUROC column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:35.728Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-021","kind":"claim","name":"Reported R² for ESM-2","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["Mutated RBD binding prediction"]},"source_ids":["prime-2026"],"links":[{"relation":"subject","target_id":"lit-021"}],"attributes":{"field":"attributes.printed_value","value":"0.0248","source_locator":"Table 1, ESM-2 8M / Mean / not fine-tuned row, Position-Stratified Split Binding R² column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.525183+00:00","notes":"Resolved model row spans and Mean/CLS subrows in JATS: selected Mean, not fine-tuned (cross), Position-Stratified Split > Binding > R-squared. Central value agrees; uncertainty is retained in evidence. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-022","kind":"claim","name":"Reported R² for ESM-C","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["Mutated RBD binding prediction"]},"source_ids":["prime-2026"],"links":[{"relation":"subject","target_id":"lit-022"}],"attributes":{"field":"attributes.printed_value","value":"-0.0162","source_locator":"Table 1, ESM-C 300M / Mean / not fine-tuned row, Position-Stratified Split Binding R² column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.526387+00:00","notes":"Resolved model row spans and Mean/CLS subrows in JATS: selected Mean, not fine-tuned (cross), Position-Stratified Split > Binding > R-squared. Central value agrees; uncertainty is retained in evidence. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-023","kind":"claim","name":"Reported Mean |Spearman rho| for PST","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["Zero-shot variant effect prediction"]},"source_ids":["pst-2025"],"links":[{"relation":"subject","target_id":"lit-023"}],"attributes":{"field":"attributes.printed_value","value":"0.501","source_locator":"Table 2, PST row, Zero-shot VEP Mean |ρ| column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.527633+00:00","notes":"Zero-shot VEP is the last metric group; selected Mean absolute rho, not GO/EC/binding-site metrics. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-024","kind":"claim","name":"Reported Mean |Spearman rho| for ESM-2","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["Zero-shot variant effect prediction"]},"source_ids":["pst-2025"],"links":[{"relation":"subject","target_id":"lit-024"}],"attributes":{"field":"attributes.printed_value","value":"0.489","source_locator":"Table 2, ESM-2 row, Zero-shot VEP Mean |ρ| column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.528656+00:00","notes":"Zero-shot VEP is the last metric group; selected Mean absolute rho, not GO/EC/binding-site metrics. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-025","kind":"claim","name":"Reported F1-Score for scGPT","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Cell-type identification"]},"source_ids":["single-cell-peft-2024"],"links":[{"relation":"subject","target_id":"lit-025"}],"attributes":{"field":"attributes.printed_value","value":"0.734","source_locator":"Table 2, M.S. / scGPT row, F1-Score column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.530269+00:00","notes":"Selected M.S. dataset block, first scGPT/Geneformer occurrences. F1-Score is last column; later dataset blocks deliberately excluded. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-026","kind":"claim","name":"Reported F1-Score for Geneformer","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Cell-type identification"]},"source_ids":["single-cell-peft-2024"],"links":[{"relation":"subject","target_id":"lit-026"}],"attributes":{"field":"attributes.printed_value","value":"0.388","source_locator":"Table 2, M.S. / Geneformer row, F1-Score column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.531756+00:00","notes":"Selected M.S. dataset block, first scGPT/Geneformer occurrences. F1-Score is last column; later dataset blocks deliberately excluded. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-027","kind":"claim","name":"Reported Partial-label accuracy for C2S (GPT-2 Large)","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Combinatorial cell-label classification"]},"source_ids":["cell2sentence-2024"],"links":[{"relation":"subject","target_id":"lit-027"}],"attributes":{"field":"attributes.printed_value","value":"0.631","source_locator":"Table 3, Partial label / C2S (GPT-2 Large) row, L1000 Acc column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.533640+00:00","notes":"Read inline small-caps/bold XML in document order, restoring Geneformer and GPT-2 Large labels. Selected Partial label (first block), L1000 > Acc, not AUROC or Full label. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-028","kind":"claim","name":"Reported Partial-label accuracy for Geneformer","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Combinatorial cell-label classification"]},"source_ids":["cell2sentence-2024"],"links":[{"relation":"subject","target_id":"lit-028"}],"attributes":{"field":"attributes.printed_value","value":"0.419","source_locator":"Table 3, Partial label / Geneformer row, L1000 Acc column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.535220+00:00","notes":"Read inline small-caps/bold XML in document order, restoring Geneformer and GPT-2 Large labels. Selected Partial label (first block), L1000 > Acc, not AUROC or Full label. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-029","kind":"claim","name":"Reported F1 for scGPT","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Cell-type annotation"]},"source_ids":["scelmo-2025"],"links":[{"relation":"subject","target_id":"lit-029"}],"attributes":{"field":"attributes.printed_value","value":"0.550","source_locator":"Table 1, hPancreas zero-shot / scGPT (z) row, F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.537541+00:00","notes":"Selected first hPancreas zero-shot block and F1 last column. Caption says some scores are copied from GenePT; this is source checking of the reported table, not independent experimental evidence. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-030","kind":"claim","name":"Reported F1 for Geneformer","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Cell-type annotation"]},"source_ids":["scelmo-2025"],"links":[{"relation":"subject","target_id":"lit-030"}],"attributes":{"field":"attributes.printed_value","value":"0.270","source_locator":"Table 1, hPancreas zero-shot / Geneformer (z) row, F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.539630+00:00","notes":"Selected first hPancreas zero-shot block and F1 last column. Caption says some scores are copied from GenePT; this is source checking of the reported table, not independent experimental evidence. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-031","kind":"claim","name":"Reported AUROC for scRegNet (Geneformer backbone)","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Gene-regulatory link prediction"]},"source_ids":["scregnet-2025"],"links":[{"relation":"subject","target_id":"lit-031"}],"attributes":{"field":"attributes.printed_value","value":"0.89","source_locator":"Table 2, scRegNet (w/ Geneformer) row, hESC AUROC entry","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.541287+00:00","notes":"Read break elements: cells contain AUROC on first line then AUPRC. Selected hESC (first cell type), first line. Caption specifies 500 most-variable genes and 50 independent evaluations. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-032","kind":"claim","name":"Reported AUROC for scRegNet (scBERT backbone)","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Gene-regulatory link prediction"]},"source_ids":["scregnet-2025"],"links":[{"relation":"subject","target_id":"lit-032"}],"attributes":{"field":"attributes.printed_value","value":"0.88","source_locator":"Table 2, scRegNet (w/ scBERT) row, hESC AUROC entry","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.542807+00:00","notes":"Read break elements: cells contain AUROC on first line then AUPRC. Selected hESC (first cell type), first line. Caption specifies 500 most-variable genes and 50 independent evaluations. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-033","kind":"claim","name":"Reported Accuracy for ProkBERT-mini","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["E. coli sigma70 promoter prediction"]},"source_ids":["prokbert-2024"],"links":[{"relation":"subject","target_id":"lit-033"}],"attributes":{"field":"attributes.printed_value","value":"0.87","source_locator":"Table 3, ProkBERT-mini row, Accuracy column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:36.197Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-034","kind":"claim","name":"Reported Accuracy for Promotech","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["E. coli sigma70 promoter prediction"]},"source_ids":["prokbert-2024"],"links":[{"relation":"subject","target_id":"lit-034"}],"attributes":{"field":"attributes.printed_value","value":"0.71","source_locator":"Table 3, Promotech row, Accuracy column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:36.197Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-035","kind":"claim","name":"Reported Promoter-class F1 for Eco70PromBERT","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["E. coli sigma70 promoter prediction"]},"source_ids":["cyaprombert-2022"],"links":[{"relation":"subject","target_id":"lit-035"}],"attributes":{"field":"attributes.printed_value","value":"0.91","source_locator":"TABLE 3, Eco70PromBERT (BERT-base + 1bp tokenizer) row, F1 score Promoter column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.544033+00:00","notes":"F1 score is the third two-column group; selected Promoter subcolumn, not AUROC or precision. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-036","kind":"claim","name":"Reported Promoter-class F1 for iPro70-FMWin","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["E. coli sigma70 promoter prediction"]},"source_ids":["cyaprombert-2022"],"links":[{"relation":"subject","target_id":"lit-036"}],"attributes":{"field":"attributes.printed_value","value":"0.90","source_locator":"TABLE 3, iPro70-FMWin row, F1 score Promoter column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.544926+00:00","notes":"F1 score is the third two-column group; selected Promoter subcolumn, not AUROC or precision. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-037","kind":"claim","name":"Reported MCC for EVO2","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Genome-wide prophage detection"]},"source_ids":["lambda-prophage-2026"],"links":[{"relation":"subject","target_id":"lit-037"}],"attributes":{"field":"attributes.printed_value","value":"0.680","source_locator":"Table 5, EVO2 row, MCC column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:36.240Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-038","kind":"claim","name":"Reported MCC for geNomad","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Genome-wide prophage detection"]},"source_ids":["lambda-prophage-2026"],"links":[{"relation":"subject","target_id":"lit-038"}],"attributes":{"field":"attributes.printed_value","value":"0.794","source_locator":"Table 5, geNomad row, MCC column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:36.240Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-039","kind":"claim","name":"Reported F1 score for NABAS+","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Metagenomic taxonomic classification"]},"source_ids":["nabas-plus-2025"],"links":[{"relation":"subject","target_id":"lit-039"}],"attributes":{"field":"attributes.printed_value","value":"0.719","source_locator":"Table 3, Sample19-new / NABAS+ row, F1 score column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.546107+00:00","notes":"Selected Sample19-new explicitly, not Sample19-old, and F1 rather than precision/recall. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-040","kind":"claim","name":"Reported F1 score for MetaPhlAn3","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Metagenomic taxonomic classification"]},"source_ids":["nabas-plus-2025"],"links":[{"relation":"subject","target_id":"lit-040"}],"attributes":{"field":"attributes.printed_value","value":"0.753","source_locator":"Table 3, Sample19-new / MetaPhlAn3 row, F1 score column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.547023+00:00","notes":"Selected Sample19-new explicitly, not Sample19-old, and F1 rather than precision/recall. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-041","kind":"claim","name":"Reported Success rate, ligand all-atom RMSD <2 Å for Chai-1","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Lipid–protein binding pose"]},"source_ids":["lipp-2026"],"links":[{"relation":"subject","target_id":"lit-041"}],"attributes":{"field":"attributes.printed_value","value":"60.7","source_locator":"Table 2, Chai-1 row, LiPP (N=331) % Success Rate column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.548973+00:00","notes":"JATS label is bare 2, which caused original parser miss. LiPP N=331 full-set column selected, not N=36 test subset. Caption success is lipid all-atom RMSD <2 Angstrom; PB-valid is a separate table. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-042","kind":"claim","name":"Reported Success rate, ligand all-atom RMSD <2 Å for DiffDock-L","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Lipid–protein binding pose"]},"source_ids":["lipp-2026"],"links":[{"relation":"subject","target_id":"lit-042"}],"attributes":{"field":"attributes.printed_value","value":"46.8","source_locator":"Table 2, DiffDock-L row, LiPP (N=331) % Success Rate column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.550691+00:00","notes":"JATS label is bare 2, which caused original parser miss. LiPP N=331 full-set column selected, not N=36 test subset. Caption success is lipid all-atom RMSD <2 Angstrom; PB-valid is a separate table. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-043","kind":"claim","name":"Reported Forward-screening success rate for DiffDock-NMDN","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand virtual screening"]},"source_ids":["nmdn-2025"],"links":[{"relation":"subject","target_id":"lit-043"}],"attributes":{"field":"attributes.printed_value","value":"66.7","source_locator":"Table 2, DiffDock-NMDN / NMDN row, success rate (%) column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.552697+00:00","notes":"All scoring functions share DiffDock-NMDN poses via rowspan. Selected forward-screening success percentage, not docking pose success or scoring-power correlation. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-044","kind":"claim","name":"Reported Forward-screening success rate for Vina","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand virtual screening"]},"source_ids":["nmdn-2025"],"links":[{"relation":"subject","target_id":"lit-044"}],"attributes":{"field":"attributes.printed_value","value":"42.1","source_locator":"Table 2, Vina scoring row, success rate (%) column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.554518+00:00","notes":"All scoring functions share DiffDock-NMDN poses via rowspan. Selected forward-screening success percentage, not docking pose success or scoring-power correlation. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-045","kind":"claim","name":"Reported Median ligand RMSD for Boltz-1","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand pose prediction"]},"source_ids":["boltz-stereochemistry-2025"],"links":[{"relation":"subject","target_id":"lit-045"}],"attributes":{"field":"attributes.printed_value","value":"1.393","source_locator":"Table 1, Boltz-1 row, Ligand RMSD (Å) column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.555674+00:00","notes":"JATS label is bare 1. Selected Ligand RMSD column in Plinder-L95, not Protein RMSD; retained method row without refinement settings. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-046","kind":"claim","name":"Reported Median ligand RMSD for DiffDock","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand pose prediction"]},"source_ids":["boltz-stereochemistry-2025"],"links":[{"relation":"subject","target_id":"lit-046"}],"attributes":{"field":"attributes.printed_value","value":"1.342","source_locator":"Table 1, DiffDock row, Ligand RMSD (Å) column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.556572+00:00","notes":"JATS label is bare 1. Selected Ligand RMSD column in Plinder-L95, not Protein RMSD; retained method row without refinement settings. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-047","kind":"claim","name":"Reported Pearson R for Boltz-2","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Ligand potency prediction using generated poses"]},"source_ids":["mpro-pose-affinity-2025"],"links":[{"relation":"subject","target_id":"lit-047"}],"attributes":{"field":"attributes.printed_value","value":"0.800","source_locator":"Table 3, Boltz-2 row, Pearson’s R column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.557756+00:00","notes":"JATS label is bare 3. Selected SARS-CoV-2 Mpro potency Pearson R, not MERS-CoV table 2 or Boltz-2-Internal row; uncertainty retained in evidence. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-048","kind":"claim","name":"Reported Pearson R for DiffDock","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Ligand potency prediction using generated poses"]},"source_ids":["mpro-pose-affinity-2025"],"links":[{"relation":"subject","target_id":"lit-048"}],"attributes":{"field":"attributes.printed_value","value":"0.695","source_locator":"Table 3, DiffDock row, Pearson’s R column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.558815+00:00","notes":"JATS label is bare 3. Selected SARS-CoV-2 Mpro potency Pearson R, not MERS-CoV table 2 or Boltz-2-Internal row; uncertainty retained in evidence. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-b3-003","kind":"claim","name":"Reported F1 for Mouse-Geneformer","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Human thymus cell-type classification"]},"source_ids":["mouse-geneformer-2025"],"links":[{"relation":"subject","target_id":"lit-b3-003"}],"attributes":{"field":"attributes.printed_value","value":"48.57","source_locator":"Table 4, h/ Thymus row, Mouse-Geneformer Zero-shot F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.392488+00:00","notes":"h/ Thymus row, four human cell types; zero-shot model blocks use Acc then F1, not fine-tuned scores. Ortholog-based conversion evaluated, not mouse cell annotation. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-004","kind":"claim","name":"Reported F1 for Human-Geneformer","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Human thymus cell-type classification"]},"source_ids":["mouse-geneformer-2025"],"links":[{"relation":"subject","target_id":"lit-b3-004"}],"attributes":{"field":"attributes.printed_value","value":"74.48","source_locator":"Table 4, h/ Thymus row, Human-Geneformer Zero-shot F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.394073+00:00","notes":"h/ Thymus row, four human cell types; zero-shot model blocks use Acc then F1, not fine-tuned scores. Ortholog-based conversion evaluated, not mouse cell annotation. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-005","kind":"claim","name":"Reported F1 for scLLMDA","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Cross-platform scATAC cell-type annotation"]},"source_ids":["scatac-llmda-2026"],"links":[{"relation":"subject","target_id":"lit-b3-005"}],"attributes":{"field":"attributes.printed_value","value":"0.6525","source_locator":"Table 2, scLLMDA row, Ref: MosA1 / Q: WholeBrainA F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.395850+00:00","notes":"First reference/query block MosA1 to WholeBrainA, F1 second column in block; direction of transfer is part of protocol. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-006","kind":"claim","name":"Reported F1 for MINGLE","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Cross-platform scATAC cell-type annotation"]},"source_ids":["scatac-llmda-2026"],"links":[{"relation":"subject","target_id":"lit-b3-006"}],"attributes":{"field":"attributes.printed_value","value":"0.6256","source_locator":"Table 2, MINGLE row, Ref: MosA1 / Q: WholeBrainA F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.397272+00:00","notes":"First reference/query block MosA1 to WholeBrainA, F1 second column in block; direction of transfer is part of protocol. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-011","kind":"claim","name":"Reported Adjusted Rand Index for GenePT-w","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Cell-type structure in frozen embeddings"]},"source_ids":["genept-2024"],"links":[{"relation":"subject","target_id":"lit-b3-011"}],"attributes":{"field":"attributes.printed_value","value":"0.54","source_locator":"Table 2, Aorta / Cell type row, GenePT-w ARI column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.399274+00:00","notes":"First Cell type row belongs to Aorta, not preceding Phenotype row or later organs. ARI is first in each three-metric method block, not AMI/ASW. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-012","kind":"claim","name":"Reported Adjusted Rand Index for scGPT","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Cell-type structure in frozen embeddings"]},"source_ids":["genept-2024"],"links":[{"relation":"subject","target_id":"lit-b3-012"}],"attributes":{"field":"attributes.printed_value","value":"0.47","source_locator":"Table 2, Aorta / Cell type row, scGPT ARI column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.402215+00:00","notes":"First Cell type row belongs to Aorta, not preceding Phenotype row or later organs. ARI is first in each three-metric method block, not AMI/ASW. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-013","kind":"claim","name":"Reported Balanced accuracy for Best frozen single-cell foundation model","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Donor-aware age-class prediction"]},"source_ids":["single-cell-aging-probes-2026"],"links":[{"relation":"subject","target_id":"lit-b3-013"}],"attributes":{"field":"attributes.printed_value","value":"0.322","source_locator":"Table 2, AIDA v2 row, scFM BA ± SD column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:44.421Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b3-014","kind":"claim","name":"Reported Balanced accuracy for Gene-expression PCA","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Donor-aware age-class prediction"]},"source_ids":["single-cell-aging-probes-2026"],"links":[{"relation":"subject","target_id":"lit-b3-014"}],"attributes":{"field":"attributes.printed_value","value":"0.384","source_locator":"Table 2, AIDA v2 row, Gene-expr BA column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:44.421Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b3-015","kind":"claim","name":"Reported Cell-type accuracy for scaLR","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["PBMC cell-type classification"]},"source_ids":["scalr-2025"],"links":[{"relation":"subject","target_id":"lit-b3-015"}],"attributes":{"field":"attributes.printed_value","value":"0.942","source_locator":"Table 2, scaLR row, Cell type Accuracy column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.403844+00:00","notes":"PBMCs-BS all-feature/all-sample cell-type accuracy block, not cell-state accuracy or time. Footnote letters on model labels excluded from identity. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-016","kind":"claim","name":"Reported Cell-type accuracy for scVI + scANVI","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["PBMC cell-type classification"]},"source_ids":["scalr-2025"],"links":[{"relation":"subject","target_id":"lit-b3-016"}],"attributes":{"field":"attributes.printed_value","value":"0.939","source_locator":"Table 2, Svi-tools (scVI & scANVI) row, Cell type Accuracy column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.405089+00:00","notes":"PBMCs-BS all-feature/all-sample cell-type accuracy block, not cell-state accuracy or time. Footnote letters on model labels excluded from identity. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-017","kind":"claim","name":"Reported AUC for scXDR","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Cross-dataset single-cell drug response transfer"]},"source_ids":["scxdr-2026"],"links":[{"relation":"subject","target_id":"lit-b3-017"}],"attributes":{"field":"attributes.printed_value","value":"0.8248","source_locator":"Table 2, scXDR row, Scenario 2 column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:50.056Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b3-018","kind":"claim","name":"Reported AUC for scVI","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Cross-dataset single-cell drug response transfer"]},"source_ids":["scxdr-2026"],"links":[{"relation":"subject","target_id":"lit-b3-018"}],"attributes":{"field":"attributes.printed_value","value":"0.6970","source_locator":"Table 2, scVI row, Scenario 2 column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:50.056Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b3-019","kind":"claim","name":"Reported L1 abundance error for CAMMiQ","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Strain-level abundance quantification"]},"source_ids":["cammiq-2022"],"links":[{"relation":"subject","target_id":"lit-b3-019"}],"attributes":{"field":"attributes.printed_value","value":"0.0517","source_locator":"Table 5, B. L1 Err. / HumanGut-all row, CAMMiQ L1 Err. column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.408237+00:00","notes":"HumanGut-all in B. L1 Err. block (second occurrence), not strain count or C. L2 Err. Numbers are errors; lower is better. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-020","kind":"claim","name":"Reported L1 abundance error for Kraken2","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Strain-level abundance quantification"]},"source_ids":["cammiq-2022"],"links":[{"relation":"subject","target_id":"lit-b3-020"}],"attributes":{"field":"attributes.printed_value","value":"0.2841","source_locator":"Table 5, B. L1 Err. / HumanGut-all row, Kraken2 L1 Err. column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.411180+00:00","notes":"HumanGut-all in B. L1 Err. block (second occurrence), not strain count or C. L2 Err. Numbers are errors; lower is better. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-021","kind":"claim","name":"Reported Genus-level F1 for Lazypipe-nt","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Simulated metagenome virus-taxon retrieval"]},"source_ids":["lazypipe-2020"],"links":[{"relation":"subject","target_id":"lit-b3-021"}],"attributes":{"field":"attributes.printed_value","value":"0.932","source_locator":"Table 1, Lazypipe-nt / Genus row, F column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.412605+00:00","notes":"First Genus block selected using rank row span, not Species. Final F column is F score, not precision or recall. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-022","kind":"claim","name":"Reported Genus-level F1 for Kraken2","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Simulated metagenome virus-taxon retrieval"]},"source_ids":["lazypipe-2020"],"links":[{"relation":"subject","target_id":"lit-b3-022"}],"attributes":{"field":"attributes.printed_value","value":"0.627","source_locator":"Table 1, Kraken2 / Genus row, F column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.413635+00:00","notes":"First Genus block selected using rank row span, not Species. Final F column is F score, not precision or recall. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-023","kind":"claim","name":"Reported Macro F1 for NCD-gzip","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["CAMI II superkingdom read classification"]},"source_ids":["ncd-metagenomics-2026"],"links":[{"relation":"subject","target_id":"lit-b3-023"}],"attributes":{"field":"attributes.printed_value","value":"0.9804","source_locator":"Table 5, NCD Superkingdom row, F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.414806+00:00","notes":"NCD rank-specific table 5, F1 column; Superkingdom and Phylum are different classification granularities. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-024","kind":"claim","name":"Reported Macro F1 for NCD-gzip","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["CAMI II phylum read classification"]},"source_ids":["ncd-metagenomics-2026"],"links":[{"relation":"subject","target_id":"lit-b3-024"}],"attributes":{"field":"attributes.printed_value","value":"0.1263","source_locator":"Table 5, NCD Phylum row, F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.415788+00:00","notes":"NCD rank-specific table 5, F1 column; Superkingdom and Phylum are different classification granularities. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-025","kind":"claim","name":"Reported Average prophage F1 for VIBRANT","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Simulated prophage-contig detection"]},"source_ids":["viral-contig-simulation-2021"],"links":[{"relation":"subject","target_id":"lit-b3-025"}],"attributes":{"field":"attributes.printed_value","value":"0.169","source_locator":"Table 3, Vibrant row, Prophage F1 column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:50.134Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b3-026","kind":"claim","name":"Reported Average prophage F1 for VirSorter","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Simulated prophage-contig detection"]},"source_ids":["viral-contig-simulation-2021"],"links":[{"relation":"subject","target_id":"lit-b3-026"}],"attributes":{"field":"attributes.printed_value","value":"0.147","source_locator":"Table 3, VirSorter row, Prophage F1 column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:50.134Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b3-027","kind":"claim","name":"Reported F1 for GenomeOcean","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Natural vs artificial microbial genome sequence"]},"source_ids":["genomeocean-2025"],"links":[{"relation":"subject","target_id":"lit-b3-027"}],"attributes":{"field":"attributes.printed_value","value":"99.03","source_locator":"Table 2, GenomeOcean row, F1 column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:55.224Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b3-028","kind":"claim","name":"Reported F1 for DNABERT-2","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Natural vs artificial microbial genome sequence"]},"source_ids":["genomeocean-2025"],"links":[{"relation":"subject","target_id":"lit-b3-028"}],"attributes":{"field":"attributes.printed_value","value":"85.12","source_locator":"Table 2, DNABERT-2 row, F1 column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:55.224Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b3-029","kind":"claim","name":"Reported Genus-level F1 for kMetaShot","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Mock-community MAG taxonomy classification"]},"source_ids":["kmetashot-2025"],"links":[{"relation":"subject","target_id":"lit-b3-029"}],"attributes":{"field":"attributes.printed_value","value":"95.83","source_locator":"Table 2, F1-score % row, Genus kMS column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.417367+00:00","notes":"Real mock sequencing table, F1-score percentage row; Genus is final three-column block, selecting kMS or Gtk rather than Species/Strain. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-030","kind":"claim","name":"Reported Genus-level F1 for GTDB-Tk","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Mock-community MAG taxonomy classification"]},"source_ids":["kmetashot-2025"],"links":[{"relation":"subject","target_id":"lit-b3-030"}],"attributes":{"field":"attributes.printed_value","value":"89.80","source_locator":"Table 2, F1-score % row, Genus Gtk column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.418914+00:00","notes":"Real mock sequencing table, F1-score percentage row; Genus is final three-column block, selecting kMS or Gtk rather than Species/Strain. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-031","kind":"claim","name":"Reported F1 for Lemur","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Long-read taxonomic profiling"]},"source_ids":["lemur-magnet-2024"],"links":[{"relation":"subject","target_id":"lit-b3-031"}],"attributes":{"field":"attributes.printed_value","value":"0.376","source_locator":"Table 3, LOG 10% / Lemur row, F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.420493+00:00","notes":"LOG 10% first block, F1 column. Kraken 2 row inherits dataset via rowspan; not LOG 75% or abundance Spearman. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-032","kind":"claim","name":"Reported F1 for Kraken 2","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Long-read taxonomic profiling"]},"source_ids":["lemur-magnet-2024"],"links":[{"relation":"subject","target_id":"lit-b3-032"}],"attributes":{"field":"attributes.printed_value","value":"0.375","source_locator":"Table 3, LOG 10% / Kraken 2 row, F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.421888+00:00","notes":"LOG 10% first block, F1 column. Kraken 2 row inherits dataset via rowspan; not LOG 75% or abundance Spearman. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-033","kind":"claim","name":"Reported Mean AUC for iPro-MP","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Multi-species prokaryotic promoter detection"]},"source_ids":["ipromp-2025"],"links":[{"relation":"subject","target_id":"lit-b3-033"}],"attributes":{"field":"attributes.printed_value","value":"0.935","source_locator":"Table 2, iPro-MP row, AUC column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:55.361Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b3-034","kind":"claim","name":"Reported Mean AUC for Prompt","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Multi-species prokaryotic promoter detection"]},"source_ids":["ipromp-2025"],"links":[{"relation":"subject","target_id":"lit-b3-034"}],"attributes":{"field":"attributes.printed_value","value":"0.835","source_locator":"Table 2, Prompt row, AUC column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:55.361Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b3-035","kind":"claim","name":"Reported Genus macro AveP for ICCTax","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Hierarchical metagenomic taxonomy classification"]},"source_ids":["icctax-2025"],"links":[{"relation":"subject","target_id":"lit-b3-035"}],"attributes":{"field":"attributes.printed_value","value":"67.20","source_locator":"Table 2, ICCTax row, Genus column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:55.373Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b3-036","kind":"claim","name":"Reported Genus macro AveP for Kraken2","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Hierarchical metagenomic taxonomy classification"]},"source_ids":["icctax-2025"],"links":[{"relation":"subject","target_id":"lit-b3-036"}],"attributes":{"field":"attributes.printed_value","value":"70.56","source_locator":"Table 2, Kraken2 row, Genus column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:55.373Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b3-037","kind":"claim","name":"Reported AUC-ROC for Chai-1","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Antibody–antigen interaction prediction using folded complexes"]},"source_ids":["antibody-flexibility-2025"],"links":[{"relation":"subject","target_id":"lit-b3-037"}],"attributes":{"field":"attributes.printed_value","value":"0.86","source_locator":"Table 5, Folded row, Chai-1 (no MSA) column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:55.400Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b3-038","kind":"claim","name":"Reported AUC-ROC for Boltz-1","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Antibody–antigen interaction prediction using folded complexes"]},"source_ids":["antibody-flexibility-2025"],"links":[{"relation":"subject","target_id":"lit-b3-038"}],"attributes":{"field":"attributes.printed_value","value":"0.85","source_locator":"Table 5, Folded row, Boltz-1 (no MSA) column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:55.400Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b3-039","kind":"claim","name":"Reported Top-1 ligand RMSD <2 Å rate for Boltz-1","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand pose prediction"]},"source_ids":["boltz1-2025"],"links":[{"relation":"subject","target_id":"lit-b3-039"}],"attributes":{"field":"attributes.printed_value","value":"0.545","source_locator":"Table 1, 3 recycling rounds / 200 steps row, L-RMSD <2Å top-1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.424864+00:00","notes":"3 recycling rounds and 200 steps; L-RMSD <2 Angstrom top-1 (last column), not oracle. Five samples generated; top-1 means highest-confidence candidate. Repeated reference rows are one evaluation, not independent experiments. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-040","kind":"claim","name":"Reported Mean CDR H3 RMSD for Ibex","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Antibody loop structure prediction"]},"source_ids":["ibex-2025"],"links":[{"relation":"subject","target_id":"lit-b3-040"}],"attributes":{"field":"attributes.printed_value","value":"2.72","source_locator":"Table 1, Antibodies / Ibex row, CDR H3 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.426811+00:00","notes":"First Antibodies block, CDR H3 mean RMSD in Angstrom; excludes later Nanobodies and TCR blocks. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-041","kind":"claim","name":"Reported Mean CDR H3 RMSD for Chai-1","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Antibody loop structure prediction"]},"source_ids":["ibex-2025"],"links":[{"relation":"subject","target_id":"lit-b3-041"}],"attributes":{"field":"attributes.printed_value","value":"2.65","source_locator":"Table 1, Antibodies / Chai-1 row, CDR H3 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.428536+00:00","notes":"First Antibodies block, CDR H3 mean RMSD in Angstrom; excludes later Nanobodies and TCR blocks. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-042","kind":"claim","name":"Reported Pearson R for DEELIG","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand binding affinity prediction"]},"source_ids":["deelig-2021"],"links":[{"relation":"subject","target_id":"lit-b3-042"}],"attributes":{"field":"attributes.printed_value","value":"0.889","source_locator":"Table 2, DEELIG row, PDBbind v2016 column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:55.586Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b3-043","kind":"claim","name":"Reported Pearson R for TOPBP (Complex)","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand binding affinity prediction"]},"source_ids":["deelig-2021"],"links":[{"relation":"subject","target_id":"lit-b3-043"}],"attributes":{"field":"attributes.printed_value","value":"0.861","source_locator":"Table 2, TOPBP (Complex) row, PDBbind v2016 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.429669+00:00","notes":"TOPBP Complex reference row; PDBbind v2016 core-set Pearson correlation. Third-party comparator with cited reference; do not infer an independent new run from table inclusion. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-044","kind":"claim","name":"Reported RMSD ≤1 Å and PB-valid success for MolAS","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Physically valid protein–ligand pose selection"]},"source_ids":["molas-2026"],"links":[{"relation":"subject","target_id":"lit-b3-044"}],"attributes":{"field":"attributes.printed_value","value":"36.69","source_locator":"Table 3, PoseBusters / Mixed / AutoDock row, MolAS success column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.432565+00:00","notes":"PoseBusters, Mixed, AutoDock row within jointly trained with/without relaxation block. Selected RMSD <=1 Angstrom AND PB-valid group; five-fold average success percentage, not <=2 Angstrom. Inline bold digit nodes joined in original order. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-045","kind":"claim","name":"Reported RMSD ≤1 Å and PB-valid success for Single best solver","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Physically valid protein–ligand pose selection"]},"source_ids":["molas-2026"],"links":[{"relation":"subject","target_id":"lit-b3-045"}],"attributes":{"field":"attributes.printed_value","value":"34.34","source_locator":"Table 3, PoseBusters / Mixed / AutoDock row, SBS success column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.435454+00:00","notes":"PoseBusters, Mixed, AutoDock row within jointly trained with/without relaxation block. Selected RMSD <=1 Angstrom AND PB-valid group; five-fold average success percentage, not <=2 Angstrom. Inline bold digit nodes joined in original order. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-046","kind":"claim","name":"Reported Docked frames best-matched RMSD <3 Å for AutoDock Vina holo","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Intrinsically disordered protein ensemble docking"]},"source_ids":["ensemble-idp-docking-2025"],"links":[{"relation":"subject","target_id":"lit-b3-046"}],"attributes":{"field":"attributes.printed_value","value":"27.96","source_locator":"Table 2, Ligand 47 row, AutoDock Vina Holo Docking column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:56.275Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b3-047","kind":"claim","name":"Reported Docked frames best-matched RMSD <3 Å for DiffDock holo","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Intrinsically disordered protein ensemble docking"]},"source_ids":["ensemble-idp-docking-2025"],"links":[{"relation":"subject","target_id":"lit-b3-047"}],"attributes":{"field":"attributes.printed_value","value":"21.32","source_locator":"Table 2, Ligand 47 row, DiffDock Holo Docking column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:56.275Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b3-048","kind":"claim","name":"Reported Pearson R for AK-score-ensemble","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand binding affinity scoring"]},"source_ids":["akscore-2020"],"links":[{"relation":"subject","target_id":"lit-b3-048"}],"attributes":{"field":"attributes.printed_value","value":"0.812","source_locator":"Table 2, AK-score-ensemble / learning rate 0.0007 row, Scoring Pearson (R) column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.436853+00:00","notes":"CASF-2016 scoring Pearson R with learning rate0.0007. Single-model versus ensemble blocks kept distinct; ranking/docking scores not substituted. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-049","kind":"claim","name":"Reported Pearson R for AK-score-single","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand binding affinity scoring"]},"source_ids":["akscore-2020"],"links":[{"relation":"subject","target_id":"lit-b3-049"}],"attributes":{"field":"attributes.printed_value","value":"0.759","source_locator":"Table 2, AK-score-single / learning rate 0.0007 row, Scoring Pearson (R) column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.437894+00:00","notes":"CASF-2016 scoring Pearson R with learning rate0.0007. Single-model versus ensemble blocks kept distinct; ranking/docking scores not substituted. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-050","kind":"claim","name":"Reported Pearson R for PMF + ECFP + PF (LightGBM)","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand binding energy prediction"]},"source_ids":["fingerprint-scoring-2022"],"links":[{"relation":"subject","target_id":"lit-b3-050"}],"attributes":{"field":"attributes.printed_value","value":"0.79","source_locator":"Table 1, PMF + ECFP + PF / LightGBM row, R column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.439460+00:00","notes":"Resolved two-row model rowspan: last LightGBM row is PMF+ECFP+PF; first LASSO row is PMF. Pearson R, not RMSE. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-051","kind":"claim","name":"Reported Pearson R for PMF (LASSO)","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand binding energy prediction"]},"source_ids":["fingerprint-scoring-2022"],"links":[{"relation":"subject","target_id":"lit-b3-051"}],"attributes":{"field":"attributes.printed_value","value":"0.67","source_locator":"Table 1, PMF / LASSO row, R column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.440812+00:00","notes":"Resolved two-row model rowspan: last LightGBM row is PMF+ECFP+PF; first LASSO row is PMF. Pearson R, not RMSE. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b4-001","kind":"claim","name":"Reported AUROC for ARSENAL+ChromBPNet","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["regulatory-variant scoring"]},"source_ids":["arsenal-regulatory-dna-2026"],"links":[{"relation":"subject","target_id":"lit-b4-001"}],"attributes":{"field":"attributes.printed_value","value":"0.896","source_locator":"Table 1, Yoruban LCL dsQTLs section, ARSENAL+ChromBPNet row, AUROC column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"Yoruban LCL dsQTLs; ARSENAL+ChromBPNet AUROC Transcription verified; experimental claims not independently reproduced."}}} {"id":"claim-lit-b4-002","kind":"claim","name":"Reported AUROC for PlantCAD2","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["cross-species conservation prediction"]},"source_ids":["plantcad2-2025"],"links":[{"relation":"subject","target_id":"lit-b4-002"}],"attributes":{"field":"attributes.printed_value","value":"0.725","source_locator":"Table 1, Cross-species evolutionary conservation > Conservation within Andropogoneae (Genome-wide) row, PlantCAD2 AUROC entry","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"First comparison entry is PlantCAD2; AUROC is 0.725 versus comparator 0.691. Transcription verified; experimental claims not independently reproduced."}}} {"id":"claim-lit-b4-003","kind":"claim","name":"Reported accuracy for Stacking-Auto","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["enhancer prediction"]},"source_ids":["hi-enhancer-2025"],"links":[{"relation":"subject","target_id":"lit-b4-003"}],"attributes":{"field":"attributes.printed_value","value":"80.50","source_locator":"Table 2, Ours (Stacking-Auto) row, Accuracy column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"Ours row, Accuracy column; original source method is the Stacking-Auto stage. Transcription verified; experimental claims not independently reproduced."}}} {"id":"claim-lit-b4-004","kind":"claim","name":"Reported AUROC for position-aware CNN","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["enhancer prediction"]},"source_ids":["enhancer-position-encoding-2024"],"links":[{"relation":"subject","target_id":"lit-b4-004"}],"attributes":{"field":"attributes.printed_value","value":"0.94","source_locator":"Table 2, Human section, CNN row, AUC column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"Human dataset row, CNN, AUC column. Transcription verified; experimental claims not independently reproduced."}}} {"id":"claim-lit-b4-005","kind":"claim","name":"Reported F1 for ADAR-GPT continual","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["A-to-I RNA editing site prediction"]},"source_ids":["adar-gpt-editing-2026"],"links":[{"relation":"subject","target_id":"lit-b4-005"}],"attributes":{"field":"attributes.printed_value","value":"0.763","source_locator":"Table 2, Adar-GPT (continual) row, F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"Adar-GPT continual row; XML inline decimal reordered by parser, original text verified separately. Transcription verified; experimental claims not independently reproduced."}}} {"id":"claim-lit-b4-006","kind":"claim","name":"Reported sequence recovery for R3Design","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA sequence design"]},"source_ids":["r3design-2025"],"links":[{"relation":"subject","target_id":"lit-b4-006"}],"attributes":{"field":"attributes.printed_value","value":"43.27","source_locator":"Table 3, R3Design row, Recovery (%) > Rfam column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"R3Design row, first Recovery column Rfam; 43.27 plus/minus0.56. Transcription verified; experimental claims not independently reproduced."}}} {"id":"claim-lit-b4-007","kind":"claim","name":"Reported AUROC for CUPID Data-aug-Avg","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["non-coding RNA pairwise interaction prediction"]},"source_ids":["cupid-rna-interactions-2026"],"links":[{"relation":"subject","target_id":"lit-b4-007"}],"attributes":{"field":"attributes.printed_value","value":"0.919","source_locator":"Table 1, CUPID > Data-aug-Avg row, AUROC column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"CUPID section Data-aug-Avg row; AUROC not AUPRC. Transcription verified; experimental claims not independently reproduced."}}} {"id":"claim-lit-b4-008","kind":"claim","name":"Reported AUROC for ProteinBERT LLM-encoding model","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["mRNA-protein interaction prediction"]},"source_ids":["mrna-protein-diversity-2026"],"links":[{"relation":"subject","target_id":"lit-b4-008"}],"attributes":{"field":"attributes.printed_value","value":"71.5","source_locator":"Table 2, RBP-aware test set row, auROC (%) column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:57.257Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b4-009","kind":"claim","name":"Reported AUROC for ESM2 650M","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["human-versus-viral protein classification"]},"source_ids":["viral-immune-mimicry-2025"],"links":[{"relation":"subject","target_id":"lit-b4-009"}],"attributes":{"field":"attributes.printed_value","value":"99.67","source_locator":"Table 1, ESM2 650M row, AUC (%) column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:57.274Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b4-010","kind":"claim","name":"Reported AUROC for ProtT5 embeddings + ensemble classifier","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["protein-protein binding-site prediction"]},"source_ids":["protein-binding-sites-2023"],"links":[{"relation":"subject","target_id":"lit-b4-010"}],"attributes":{"field":"attributes.printed_value","value":"0.810","source_locator":"Table 2, Dset_448 section, ProtT5 row, AUROC column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"Dset_448 block; ProtT5 AUROC, downstream ensemble retained in protocol. Transcription verified; experimental claims not independently reproduced."}}} {"id":"claim-lit-b4-011","kind":"claim","name":"Reported AUROC for CLAPE-SMB with ESM-2","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["protein-small molecule binding-site prediction"]},"source_ids":["clape-smb-2024"],"links":[{"relation":"subject","target_id":"lit-b4-011"}],"attributes":{"field":"attributes.printed_value","value":"0.917","source_locator":"Table 5, ESM-2 / SJC row, AUROC column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"ESM-2 on SJC AUROC. Transcription verified; experimental claims not independently reproduced."}}} {"id":"claim-lit-b4-012","kind":"claim","name":"Reported AUPRC for Vaxign-DL + ESM","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["vaccine-antigen candidate prediction"]},"source_ids":["vaxign-esm-2024"],"links":[{"relation":"subject","target_id":"lit-b4-012"}],"attributes":{"field":"attributes.printed_value","value":"0.92","source_locator":"Table 2, 4 Layers row, AUPRC column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"Source spells 4 Layerss; AUPRC0.92±0.013. Transcription verified; experimental claims not independently reproduced."}}} {"id":"claim-lit-b4-013","kind":"claim","name":"Reported AUROC for scGPT + residual geometry","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["gene-regulatory signal prediction"]},"source_ids":["single-cell-residual-geometry-2026"],"links":[{"relation":"subject","target_id":"lit-b4-013"}],"attributes":{"field":"attributes.printed_value","value":"0.677","source_locator":"Table 4, Immune row, scGPT > +geom AUROC column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"Immune row, scGPT +geom (second numeric column), not Geneformer or delta. Transcription verified; experimental claims not independently reproduced."}}} {"id":"claim-lit-b4-014","kind":"claim","name":"Reported F1 for GREmLN","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["cell-type annotation"]},"source_ids":["gremln-2026"],"links":[{"relation":"subject","target_id":"lit-b4-014"}],"attributes":{"field":"attributes.printed_value","value":"0.937","source_locator":"Table 2, Cell type annotation(zero-shot), Non-immune cells, F1 row, GREmLN column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:57.502Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b4-015","kind":"claim","name":"Reported F1 for Cell-DINO ViT-L","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["protein localization classification"]},"source_ids":["cell-dino-2025"],"links":[{"relation":"subject","target_id":"lit-b4-015"}],"attributes":{"field":"attributes.printed_value","value":"65.5","source_locator":"Table 2, HPA-FoV section, Cell-DINO row, PL column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"HPA-FoV Cell-DINO PL column (protein localisation), not CL. Transcription verified; experimental claims not independently reproduced."}}} {"id":"claim-lit-b4-016","kind":"claim","name":"Reported precision at 50% recall for scGen","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["differentially expressed gene identification"]},"source_ids":["insilico-perturbation-auprc-2025"],"links":[{"relation":"subject","target_id":"lit-b4-016"}],"attributes":{"field":"attributes.printed_value","value":"0.91","source_locator":"Table 3, CD14+Mono section, scGen row, Precision at 50% Recall column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"CD14+Mono scGen; precision at50%recall. Transcription verified; experimental claims not independently reproduced."}}} {"id":"claim-lit-b4-017","kind":"claim","name":"Reported F1 for TCINet + HTRS","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["pathogen detection"]},"source_ids":["metagenomic-pathogens-2025"],"links":[{"relation":"subject","target_id":"lit-b4-017"}],"attributes":{"field":"attributes.printed_value","value":"0.84","source_locator":"Table 3, MetaHIT dataset section, TCINet + HTRS (Ours) row, F1-score column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"MetaHIT block TCINet+HTRS F1-score. Transcription verified; experimental claims not independently reproduced."}}} {"id":"claim-lit-b4-018","kind":"claim","name":"Reported accuracy for DETIRE","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["viral sequence detection"]},"source_ids":["detire-viral-metagenomes-2023"],"links":[{"relation":"subject","target_id":"lit-b4-018"}],"attributes":{"field":"attributes.printed_value","value":"0.8772","source_locator":"Table 1, Accuracy row, DETIRE column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:58.392Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b4-019","kind":"claim","name":"Reported accuracy for PC-mer + LR","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["metagenomic genus classification"]},"source_ids":["pc-mer-2024"],"links":[{"relation":"subject","target_id":"lit-b4-019"}],"attributes":{"field":"attributes.printed_value","value":"96.95","source_locator":"Table 3, AMP section, PC-mer + LR k=8 row, Accuracy (%) column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"AMP block PC-mer+LR section, k=8, first numeric value after k is Accuracy. Transcription verified; experimental claims not independently reproduced."}}} {"id":"claim-lit-b4-020","kind":"claim","name":"Reported accuracy for MDL4Microbiome","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["microbiome disease-state classification"]},"source_ids":["mdl4microbiome-2022"],"links":[{"relation":"subject","target_id":"lit-b4-020"}],"attributes":{"field":"attributes.printed_value","value":"0.97","source_locator":"Table 3, CRC row, MDL4Microbiome column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:58.492Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b4-021","kind":"claim","name":"Reported Pearson correlation for binding-affinity meta-model","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["protein-ligand binding affinity prediction"]},"source_ids":["ligand-affinity-meta-model-2024"],"links":[{"relation":"subject","target_id":"lit-b4-021"}],"attributes":{"field":"attributes.printed_value","value":"0.777","source_locator":"Table 4, Meta-models row, CASF-2016 Benchmark > PCC column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"Meta-models CASF-2016 PCC. Confirmed XML training-set rowspan inherits preceding row, so0.777 maps to PCC. Transcription verified; experimental claims not independently reproduced."}}} {"id":"claim-lit-b4-022","kind":"claim","name":"Reported AUROC for DeepInterAware","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["antigen-antibody HIV neutralization prediction"]},"source_ids":["deepinteraware-2025"],"links":[{"relation":"subject","target_id":"lit-b4-022"}],"attributes":{"field":"attributes.printed_value","value":"0.826","source_locator":"Table 2, Ab Unseen section, DeepInterAware row, AUROC column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"Ab Unseen block DeepInterAware AUROC0.826±0.017, not Ag Unseen. Transcription verified; experimental claims not independently reproduced."}}} {"id":"claim-lit-b4-023","kind":"claim","name":"Reported AUROC for TransBind","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["transcription-factor DNA binding-site prediction"]},"source_ids":["transbind-2026"],"links":[{"relation":"subject","target_id":"lit-b4-023"}],"attributes":{"field":"attributes.printed_value","value":"0.9508","source_locator":"Table 2, TransBind row, AUROC column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:58.585Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b4-024","kind":"claim","name":"Reported AUROC for ESM2_AMPS","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["protein-protein interaction prediction"]},"source_ids":["esm2-amp-2025"],"links":[{"relation":"subject","target_id":"lit-b4-024"}],"attributes":{"field":"attributes.printed_value","value":"0.68","source_locator":"Table 4, ESM2_AMPS row, AUROC column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:58.625Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"clape-smb-2024","kind":"source","name":"Protein-small molecule binding site prediction based on a pre-trained protein language model with contrastive learning","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["proteins-complexes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11542454/","version":"version of record","retrieved_at":"2026-09-15T23:37:05Z","doi":"10.1186/s13321-024-00920-2","publication_status":"peer_reviewed","year":2024,"artifact_sha256":"215919244c3dd2dfb0b55fce91c211430fd8d4aee4bb28bd03eab9f4feb73e62","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11542454/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:06Z","legacy_paper":{"id":"clape-smb-2024","title":"Protein-small molecule binding site prediction based on a pre-trained protein language model with contrastive learning","year":2024,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11542454/","primary_domain":"proteins-complexes","retrieved_utc":"2026-09-15T23:37:05Z","notes":"Primary full text verified via Europe PMC fullTextXML; venue: Journal of Cheminformatics; PMC ID: PMC11542454. ESM-2 feature extractor embedded in CLAPE-SMB; score belongs to combined downstream system.","doi":"10.1186/s13321-024-00920-2"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"clathrin-plm-2025","kind":"source","name":"Advancing the accuracy of clathrin protein prediction through multi-source protein language models","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["proteins-complexes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12238356/","version":"journal full text in PMC","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.1038/s41598-025-08510-4","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"2edc86b25707c1b737d26117093ce8d856e79cc5d0b335f27c1c341f887f1c7e","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12238356/fullTextXML","artifact_retrieved_at":"2026-09-16T10:38:57.558194+00:00","legacy_paper":{"id":"clathrin-plm-2025","title":"Advancing the accuracy of clathrin protein prediction through multi-source protein language models","year":2025,"publication_status":"peer_reviewed","version":"journal full text in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12238356/","primary_domain":"proteins-complexes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.1038/s41598-025-08510-4","notes":"Numeric result checked against Table 2 in primary full-text XML; journal/source: Scientific Reports."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"cobra-rna-binding-2026","kind":"source","name":"CoBRA: compound binding site prediction using RNA language model","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["rna-transcriptomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12790621/","version":"journal full text in PMC","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.1093/bib/bbaf713","publication_status":"peer_reviewed","year":2026,"artifact_sha256":"8c6a6f00f5fa5f62acf301a66e9e6fa9ef11c7a05ad9b7447d2ade2ce8eba793","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12790621/fullTextXML","artifact_retrieved_at":"2026-09-16T10:38:57.558197+00:00","legacy_paper":{"id":"cobra-rna-binding-2026","title":"CoBRA: compound binding site prediction using RNA language model","year":2026,"publication_status":"peer_reviewed","version":"journal full text in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12790621/","primary_domain":"rna-transcriptomes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.1093/bib/bbaf713","notes":"Numeric result checked against Table 2 in primary full-text XML; journal/source: Briefings in Bioinformatics."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"codonbert-vaccines-2024","kind":"source","name":"CodonBERT large language model for mRNA vaccines","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["rna-transcriptomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11368176/","version":"journal full text in PMC","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.1101/gr.278870.123","publication_status":"peer_reviewed","year":2024,"artifact_sha256":"2968073753e6d44feff9c08b131edf23145e95b171434539dddf77bb92847033","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11368176/fullTextXML","artifact_retrieved_at":"2026-09-16T10:38:57.558201+00:00","legacy_paper":{"id":"codonbert-vaccines-2024","title":"CodonBERT large language model for mRNA vaccines","year":2024,"publication_status":"peer_reviewed","version":"journal full text in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11368176/","primary_domain":"rna-transcriptomes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.1101/gr.278870.123","notes":"Numeric result checked against Table 2. in primary full-text XML; journal/source: Genome Research."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"cupid-rna-interactions-2026","kind":"source","name":"Computational understanding of non-coding RNA pairwise interactions","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["rna-transcriptomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12957212/","version":"PMC archival version PMC12957212.1","retrieved_at":"2026-09-15T23:37:05Z","doi":"10.3389/frai.2026.1749205","publication_status":"peer_reviewed","year":2026,"artifact_sha256":"0e6719410b390ee9c4858bb9321042851100fb74df3aa109bf2af2b8aaff7ac1","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12957212/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:06Z","legacy_paper":{"id":"cupid-rna-interactions-2026","title":"Computational understanding of non-coding RNA pairwise interactions","year":2026,"publication_status":"peer_reviewed","version":"PMC archival version PMC12957212.1","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12957212/","primary_domain":"rna-transcriptomes","retrieved_utc":"2026-09-15T23:37:05Z","notes":"Primary full text verified via Europe PMC fullTextXML; venue: Frontiers in Artificial Intelligence; PMC ID: PMC12957212. RNA-RNA pairwise interaction predictor; not a foundation model.","doi":"10.3389/frai.2026.1749205"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"cyaprombert-2022","kind":"source","name":"TSSNote-CyaPromBERT: Development of an integrated platform for highly accurate promoter prediction and visualization of Synechococcus sp. and Synechocystis sp. through a state-of-the-art natural language processing model BERT","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC9745317/","version":"version of record","retrieved_at":"2026-09-15T23:25:00Z","doi":"10.3389/fgene.2022.1067562","publication_status":"peer_reviewed","year":2022,"artifact_sha256":"74278ccd77b2bc00a3f4434546545e8bdec8b0652a0e5d1862ec0f91decccd8d","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC9745317/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:16.544033+00:00","legacy_paper":{"id":"cyaprombert-2022","title":"TSSNote-CyaPromBERT: Development of an integrated platform for highly accurate promoter prediction and visualization of Synechococcus sp. and Synechocystis sp. through a state-of-the-art natural language processing model BERT","year":2022,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC9745317/","primary_domain":"microbes-communities","retrieved_utc":"2026-09-15T23:25:00Z","notes":"Primary full text via Europe PMC XML; venue: Frontiers in Genetics; PMC ID: PMC9745317.","doi":"10.3389/fgene.2022.1067562"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"dart-eval-regulatory-2024","kind":"source","name":"DART-Eval: A Comprehensive DNA Language Model Evaluation Benchmark on Regulatory DNA","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["dna-genomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://proceedings.neurips.cc/paper_files/paper/2024/file/71998bfc3217ffe1cca1ee084dfadadd-Paper-Datasets_and_Benchmarks_Track.pdf","version":"NeurIPS 2024 Datasets and Benchmarks Track proceedings","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.52202/079017-1981","publication_status":"peer_reviewed","year":2024,"artifact_sha256":"e5aee5b1f7cc6fd961b1d2a131d02cf243b79e091d5e418fbabee7fde9b39b22","artifact_url":"https://proceedings.neurips.cc/paper_files/paper/2024/file/71998bfc3217ffe1cca1ee084dfadadd-Paper-Datasets_and_Benchmarks_Track.pdf","artifact_retrieved_at":"2026-09-16T10:38:57.558203+00:00","legacy_paper":{"id":"dart-eval-regulatory-2024","title":"DART-Eval: A Comprehensive DNA Language Model Evaluation Benchmark on Regulatory DNA","year":2024,"publication_status":"peer_reviewed","version":"NeurIPS 2024 Datasets and Benchmarks Track proceedings","source_url":"https://proceedings.neurips.cc/paper_files/paper/2024/file/71998bfc3217ffe1cca1ee084dfadadd-Paper-Datasets_and_Benchmarks_Track.pdf","primary_domain":"dna-genomes","retrieved_utc":"2026-09-15T23:33:26Z","notes":"Proceedings Table 3, DNABERT-2 Zero-Shot Accuracy 0.876 checked directly; the PMC/arXiv manuscript carries the same printed row.","doi":"10.52202/079017-1981"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"debfold-2024","kind":"source","name":"DEBFold: Computational Identification of RNA Secondary Structures for Sequences across Structural Families Using Deep Learning","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["rna-transcriptomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11094721/","version":"PMC11094721.1","retrieved_at":"2026-09-15T23:25:00Z","doi":"10.1021/acs.jcim.4c00458","publication_status":"peer_reviewed","year":2024,"artifact_sha256":"e8f960eafb7f00edfdd81d4fb75c6de838e9b872b7e18875fc7a5bff2a2f72b3","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11094721/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:16.509290+00:00","legacy_paper":{"id":"debfold-2024","title":"DEBFold: Computational Identification of RNA Secondary Structures for Sequences across Structural Families Using Deep Learning","year":2024,"publication_status":"peer_reviewed","version":"PMC11094721.1","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11094721/","primary_domain":"rna-transcriptomes","retrieved_utc":"2026-09-15T23:25:00Z","notes":"Primary full text via Europe PMC XML; venue: Journal of Chemical Information and Modeling; PMC ID: PMC11094721.","doi":"10.1021/acs.jcim.4c00458"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"deelig-2021","kind":"source","name":"DEELIG: A Deep Learning Approach to Predict Protein-Ligand Binding Affinity","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC8274096/","version":"PMC archival version PMC8274096.1","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.1177/11779322211030364","publication_status":"peer_reviewed","year":2021,"artifact_sha256":"5a7620c18d0622561004e1e25b5cfaf7399e93df3547eeefdd4cf6d300bb8aba","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC8274096/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:55.586Z","legacy_paper":{"id":"deelig-2021","title":"DEELIG: A Deep Learning Approach to Predict Protein-Ligand Binding Affinity","year":2021,"publication_status":"peer_reviewed","version":"PMC archival version PMC8274096.1","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC8274096/","primary_domain":"molecular-interactions","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: Bioinformatics and Biology Insights; PMC ID: PMC8274096.","doi":"10.1177/11779322211030364"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"deepinteraware-2025","kind":"source","name":"DeepInterAware: Deep Interaction Interface‐Aware Network for Improving Antigen‐Antibody Interaction Prediction from Sequence Data","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11967782/","version":"version of record","retrieved_at":"2026-09-15T23:37:05Z","doi":"10.1002/advs.202412533","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"25d3561934965f754d8712ec02b2052e9a3979e433b88ecebd5db14e930f17a1","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11967782/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:06Z","legacy_paper":{"id":"deepinteraware-2025","title":"DeepInterAware: Deep Interaction Interface‐Aware Network for Improving Antigen‐Antibody Interaction Prediction from Sequence Data","year":2025,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11967782/","primary_domain":"molecular-interactions","retrieved_utc":"2026-09-15T23:37:05Z","notes":"Primary full text verified via Europe PMC fullTextXML; venue: Advanced Science; PMC ID: PMC11967782. Neutralization prediction, not generic binding affinity; uncertainty printed in source table.","doi":"10.1002/advs.202412533"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"detire-viral-metagenomes-2023","kind":"source","name":"DETIRE: a hybrid deep learning model for identifying viral sequences from metagenomes","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC10313334/","version":"PMC archival version PMC10313334.1","retrieved_at":"2026-09-15T23:37:05Z","doi":"10.3389/fmicb.2023.1169791","publication_status":"peer_reviewed","year":2023,"artifact_sha256":"9ff7d32758620f7b0b0628425f62abff103ca2e33269ce3763383584bcebfc3c","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC10313334/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:58.392Z","legacy_paper":{"id":"detire-viral-metagenomes-2023","title":"DETIRE: a hybrid deep learning model for identifying viral sequences from metagenomes","year":2023,"publication_status":"peer_reviewed","version":"PMC archival version PMC10313334.1","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC10313334/","primary_domain":"microbes-communities","retrieved_utc":"2026-09-15T23:37:05Z","notes":"Primary full text verified via Europe PMC fullTextXML; venue: Frontiers in Microbiology; PMC ID: PMC10313334. Task-specific viral classifier, included as a microbial metagenomics benchmark.","doi":"10.3389/fmicb.2023.1169791"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"attributes":{"applicability":"proposed","baseline_type":"null-control","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Matched control cells, normalization and evaluation gene set.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["single-cell"]},"id":"discovery-baseline-cell-perturbation-no-change","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-perturbench"}],"name":"Cell perturbation no-change","source_ids":["src-discovery-altoslabs-perturbench"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"biological-procedure","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Receptor preparation, search box, exhaustiveness and conformers.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["molecular-interactions"]},"id":"discovery-baseline-classical-molecular-docking","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-plinder"},{"relation":"model","target_id":"discovery-model-autodock-vina"}],"name":"Classical molecular docking","source_ids":["src-discovery-ccsb-scripps-autodock-vina"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"null-control","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Matched sequence lengths and dinucleotide-preserving shuffle; fix seeds.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["genomics"]},"id":"discovery-baseline-dinucleotide-shuffled-sequence-control","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-dart-eval"}],"name":"Dinucleotide-shuffled sequence control","source_ids":["src-discovery-kundajelab-dart-eval"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"biological-procedure","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Pinned reference genomes, taxonomy and confidence setting.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["microbiome"]},"id":"discovery-baseline-exact-sequence-taxonomic-classification","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-cami-taxonomic-binning"},{"relation":"model","target_id":"discovery-model-kraken-2"}],"name":"Exact-sequence taxonomic classification","source_ids":["src-discovery-derrickwood-kraken2"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"experimental-reference","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Independent biological replicates under matching conditions; not a universal ceiling.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["single-cell"]},"id":"discovery-baseline-experimental-replicate-agreement","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-scperteval"}],"name":"Experimental replicate agreement","source_ids":["src-discovery-virtual-cell-research-community-scperteval"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"mechanistic","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Stoichiometric reconstruction, growth medium, bounds and objective.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["mechanistic-biology"]},"id":"discovery-baseline-flux-balance-prediction","kind":"baseline","links":[{"relation":"model","target_id":"discovery-model-cobrapy"}],"name":"Flux-balance prediction","source_ids":["src-discovery-opencobra-cobrapy"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"simple-statistical","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Training-only motif features and leakage-aware glycan split.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["glycomics"]},"id":"discovery-baseline-glycan-motif-feature-classifier","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-glycanml"}],"name":"Glycan motif feature classifier","source_ids":["src-discovery-bojarlab-glycowork"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"simple-statistical","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Training-only feature fitting; choose k and penalty within training folds.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["genomics"]},"id":"discovery-baseline-k-mer-ridge-regression","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-genomic-benchmarks"}],"name":"k-mer ridge regression","source_ids":["src-discovery-ml-bioinfo-ceitec-genomic-benchmarks"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"biological-procedure","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Adduct, ion mode, library version, mass tolerance and annotation resolution.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["lipidomics"]},"id":"discovery-baseline-lipid-fragmentation-library-match","kind":"baseline","links":[{"relation":"model","target_id":"discovery-model-lipidblast"}],"name":"Lipid fragmentation library match","source_ids":["src-discovery-lipidblast"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"biological-procedure","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Pinned marker database and taxonomic rank.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["microbiome"]},"id":"discovery-baseline-marker-based-microbial-profiling","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-cami-taxonomic-profiling"},{"relation":"model","target_id":"discovery-model-metaphlan"}],"name":"Marker-based microbial profiling","source_ids":["src-discovery-biobakery-metaphlan"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"biological-procedure","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Peak filtering, precursor tolerance, library and candidate set.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["metabolomics"]},"id":"discovery-baseline-mass-spectral-cosine-matching","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-massspecgym-molecule-retrieval"},{"relation":"model","target_id":"discovery-model-matchms"}],"name":"Mass spectral cosine matching","source_ids":["src-discovery-matchms-matchms"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"biological-procedure","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Pinned motif library, background frequencies and strand convention.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["genomics"]},"id":"discovery-baseline-motif-scanning","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-dart-eval"},{"relation":"model","target_id":"discovery-model-fimo"}],"name":"Motif scanning","source_ids":["src-discovery-meme"],"status":"discovered"} {"attributes":{"applicability":"source_supported","baseline_type":"simple-statistical","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Task-specific training split and predictor head.","scope_note":"One Hot is an explicitly reported comparator in official TAPE task tables. This record does not imply the same baseline protocol suits every protein task."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["protein-function"]},"id":"discovery-baseline-one-hot-protein-encoding","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-tape"}],"name":"One-hot protein encoding","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"biological-procedure","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Pinned sequence database, MSA construction and score threshold.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["protein-function"]},"id":"discovery-baseline-profile-hmm-sequence-search","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-tape-remote-homology-detection"},{"relation":"model","target_id":"discovery-model-hh-suite"}],"name":"Profile-HMM sequence search","source_ids":["src-discovery-soedinglab-hh-suite"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"biological-procedure","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Training reference set and homology leakage controls.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["protein-function"]},"id":"discovery-baseline-protein-homology-transfer","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-cafa"},{"relation":"model","target_id":"discovery-model-mmseqs2"}],"name":"Protein homology transfer","source_ids":["src-discovery-soedinglab-mmseqs2"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"learned-specialist","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Pinned backbone, checkpoint and sampling temperature.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["protein-structure"]},"id":"discovery-baseline-protein-sequence-recovery-specialist","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-proteinbench"},{"relation":"model","target_id":"discovery-model-proteinmpnn"}],"name":"Protein sequence recovery specialist","source_ids":["src-discovery-dauparas-proteinmpnn"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"null-control","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Matched candidate edge universe, edge density and seed.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["biological-networks"]},"id":"discovery-baseline-random-regulatory-network","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-beeline"}],"name":"Random regulatory network","source_ids":["src-discovery-murali-group-beeline"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"biological-procedure","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Temperature, thermodynamic parameter set and pseudoknot policy.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["rna"]},"id":"discovery-baseline-rna-minimum-free-energy-folding","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-beacon"},{"relation":"model","target_id":"discovery-model-viennarna-rnafold"}],"name":"RNA minimum-free-energy folding","source_ids":["src-discovery-viennarna-viennarna"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"simple-statistical","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Training-only GC/codon composition features and held-out split.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["rna"]},"id":"discovery-baseline-rna-sequence-composition-regression","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-mrnabench"}],"name":"RNA sequence-composition regression","source_ids":["src-discovery-morrislab-mrnabench"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"simple-statistical","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Frozen features and donor-disjoint folds; training-only regularization.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["spatial-omics"]},"id":"discovery-baseline-spatial-expression-ridge-regression","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-hest-benchmark"}],"name":"Spatial expression ridge regression","source_ids":["src-discovery-mahmoodlab-hest"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"learned-specialist","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Genome assembly, transcript context and model release.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["genomics"]},"id":"discovery-baseline-specialist-splicing-predictor","kind":"baseline","links":[{"relation":"model","target_id":"discovery-model-spliceai"}],"name":"Specialist splicing predictor","source_ids":["src-discovery-illumina-spliceai"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"simple-statistical","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Training-only expression mean with explicit perturbation averaging.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["single-cell"]},"id":"discovery-baseline-training-perturbation-mean","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-perturbench"}],"name":"Training perturbation mean","source_ids":["src-discovery-altoslabs-perturbench"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"learned-specialist","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Expression normalization, regulator list and training cells.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["biological-networks"]},"id":"discovery-baseline-tree-ensemble-regulatory-inference","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-beeline"},{"relation":"model","target_id":"discovery-model-genie3"}],"name":"Tree-ensemble regulatory inference","source_ids":["src-discovery-aertslab-genie3"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"null-control","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Same cells, preprocessing and metrics as integrated embeddings.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["single-cell"]},"id":"discovery-baseline-unintegrated-expression-reference","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-scib"}],"name":"Unintegrated expression reference","source_ids":["src-discovery-theislab-scib"],"status":"discovered"} {"attributes":{"entity_level":"evaluator","scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Genome reconstruction and taxonomic assignment evaluation","version":null,"historical_missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"AMBER assesses metagenomic genome bins and taxonomic assignments against gold-standard assignments.","summary_source_ids":["src-discovery-cami-challenge-amber"],"summary_source_locator":"Pinned README: introduction; Metrics computed per bin/per sample; input formats","sections":[{"title":"Evaluation methodology","body":"User-supplied bins plus sample-matched gold-standard assignments; example CAMI datasets are linked. The evaluator scores supplied outputs; predictor train/test partitions belong to the dataset/run being evaluated. Bin purity/completeness; sample accuracy, contamination, adjusted Rand index, binned fraction and recovered-genome counts; UniFrac for taxonomic binning. Multiple programs or parameter settings can be compared using the same reference. Uncertainty across samples, datasets or training runs must be defined by the evaluation study; this evaluator entry does not fix one experiment.","source_ids":["src-discovery-cami-challenge-amber"],"source_locator":"Pinned README: introduction; Metrics computed per bin/per sample; input formats"}],"facts":[{"label":"Datasets","value":"User-supplied bins plus sample-matched gold-standard assignments; example CAMI datasets are linked.","status":"source_checked","source_ids":["src-discovery-cami-challenge-amber"],"source_locator":"Pinned README: introduction; Metrics computed per bin/per sample; input formats"},{"label":"Splits","value":"The evaluator scores supplied outputs; predictor train/test partitions belong to the dataset/run being evaluated.","status":"inapplicable","source_ids":["src-discovery-cami-challenge-amber"],"source_locator":"Pinned README: introduction; Metrics computed per bin/per sample; input formats"},{"label":"Metrics","value":"Bin purity/completeness; sample accuracy, contamination, adjusted Rand index, binned fraction and recovered-genome counts; UniFrac for taxonomic binning.","status":"source_checked","source_ids":["src-discovery-cami-challenge-amber"],"source_locator":"Pinned README: introduction; Metrics computed per bin/per sample; input formats"},{"label":"Baselines","value":"Multiple programs or parameter settings can be compared using the same reference.","status":"source_checked","source_ids":["src-discovery-cami-challenge-amber"],"source_locator":"Pinned README: introduction; Metrics computed per bin/per sample; input formats"},{"label":"Leakage controls","value":"AMBER compares submitted bin assignments with a user-provided gold standard. It does not construct model-training partitions or certify reference-database independence; those controls belong to the evaluated study.","status":"inapplicable","source_ids":["evidence-discovery-final-amber"],"source_locator":"Implementation and benchmarking: gold-standard mapping, input formats and metrics"},{"label":"Uncertainty","value":"Uncertainty across samples, datasets or training runs must be defined by the evaluation study; this evaluator entry does not fix one experiment.","status":"inapplicable","source_ids":["src-discovery-cami-challenge-amber"],"source_locator":"Pinned README: introduction; Metrics computed per bin/per sample; input formats"},{"label":"Entity type","value":"Evaluator for genome and taxonomic binning.","status":"source_checked","source_ids":["src-discovery-cami-challenge-amber"],"source_locator":"Pinned README: introduction; Metrics computed per bin/per sample; input formats"},{"label":"Organisms","value":"The evaluator accepts any community with compatible gold-standard assignments; organism scope belongs to the input dataset.","status":"inapplicable","source_ids":["src-discovery-cami-challenge-amber"],"source_locator":"Pinned README: introduction; Metrics computed per bin/per sample; input formats"},{"label":"Assays","value":"Metagenomic sequence bins and reference assignments.","status":"source_checked","source_ids":["src-discovery-cami-challenge-amber"],"source_locator":"Pinned README: introduction; Metrics computed per bin/per sample; input formats"},{"label":"Allowed inputs","value":"Predicted sequence-to-bin assignments and sample-matched gold standards.","status":"source_checked","source_ids":["src-discovery-cami-challenge-amber"],"source_locator":"Pinned README: introduction; Metrics computed per bin/per sample; input formats"},{"label":"Adaptation","value":"AMBER scores submitted assignments; it does not prescribe predictor training.","status":"inapplicable","source_ids":["src-discovery-cami-challenge-amber"],"source_locator":"Pinned README: introduction; Metrics computed per bin/per sample; input formats"}],"strengths":[{"text":"Reports purity and completeness separately, revealing over-splitting versus contamination.","source_ids":["src-discovery-cami-challenge-amber"],"source_locator":"Pinned README: introduction; Metrics computed per bin/per sample; input formats"}],"limitations":[{"text":"Results depend on the supplied gold-standard assignments and taxonomy version. AMBER evaluation alone does not establish independence of the predictor from those references.","source_ids":["evidence-discovery-final-amber"],"source_locator":"Implementation and benchmarking: gold-standard mapping, input formats and metrics"}],"diagram":{"title":"Evaluation procedure","steps":["Allowed inputs: Predicted sequence-to-bin assignments and sample-matched gold standards.","Splits: The evaluator scores supplied outputs; predictor train/test partitions belong to the dataset/run being evaluated.","Metrics: Bin purity/completeness; sample accuracy, contamination, adjusted Rand index, binned fraction and recovered-genome counts; UniFrac for taxonomic binning."],"caption":"Conceptual procedure. Task variants and protocol versions retain their separate scoring conditions.","source_ids":["src-discovery-cami-challenge-amber"],"source_locator":"Pinned README: introduction; Metrics computed per bin/per sample; input formats"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary paper and/or task implementation reviewed for the explicitly cited methodology claims. Scope-limited absence is recorded only after the documented source search; no model runs or independent reproduction."}}},"description":"Genome reconstruction and taxonomic assignment evaluation","facets":{"areas":["microbiome"]},"id":"discovery-benchmark-amber","kind":"benchmark","links":[],"name":"AMBER","source_ids":["src-discovery-cami-challenge-amber"],"status":"discovered"} {"attributes":{"entity_level":"suite","scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Three-dimensional molecular learning tasks","version":null,"historical_missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"ATOM3D provides molecular-structure datasets and utilities for task-specific evaluation.","summary_source_ids":["src-discovery-drorlab-atom3d"],"summary_source_locator":"Pinned README: Overview; dataset access; supported formats and splitting/filtering utilities","sections":[{"title":"Evaluation methodology","body":"ATOM3D evaluates predictions from molecular structures using eight separate tasks. It supplies task-specific reference labels and partitions, ranging from random small molecules to held-out protein families and future structure-prediction targets. Each task has its own metric and representation-matched baseline; the suite is not a single universal structure score.","source_ids":["evidence-discovery-final-atom3d"],"source_locator":"Sections 3.1–3.8, 4–5; Appendix D–F; Table 8"}],"facts":[{"label":"Datasets","value":"Three-dimensional molecular data with associated labels/metadata; ligand-binding affinity is one documented example.","status":"source_checked","source_ids":["src-discovery-drorlab-atom3d"],"source_locator":"Pinned README: Overview; dataset access; supported formats and splitting/filtering utilities"},{"label":"Splits","value":"Task-specific: random molecules for SMP; 30% protein sequence identity for PIP, MSP and the strict LBA split; CATH topology groups for RES; protein targets for LEP; competition years for PSR and RSR. LBA also provides a less restrictive 60% identity split.","status":"source_checked","source_ids":["evidence-discovery-final-atom3d"],"source_locator":"Sections 3.1–3.8, 4–5; Appendix D–F; Table 8"},{"label":"Metrics","value":"SMP: MAE; PIP/MSP/LEP: classification AUROC; RES: accuracy; LBA: RMSE and correlations; PSR/RSR: correlations of predicted structure quality with reference GDT_TS/RMSD. Metrics and aggregation belong to each task.","status":"source_checked","source_ids":["evidence-discovery-final-atom3d"],"source_locator":"Sections 3.1–3.8, 4–5; Appendix D–F; Table 8"},{"label":"Baselines","value":"3D CNNs, graph networks and equivariant networks are compared with task-specific 1D/2D methods; structure-ranking tasks also use established 3D methods. Appendix F documents the per-task comparators.","status":"source_checked","source_ids":["evidence-discovery-final-atom3d"],"source_locator":"Sections 3.1–3.8, 4–5; Appendix D–F; Table 8"},{"label":"Leakage controls","value":"Protein-sequence, topology, target and temporal partitions reduce task-specific overlap; PIP prunes DIPS proteins against DB5. SMP uses a random molecular split, so it is not a scaffold-held-out test.","status":"source_checked","source_ids":["evidence-discovery-final-atom3d"],"source_locator":"Sections 3.1–3.8, 4–5; Appendix D–F; Table 8"},{"label":"Uncertainty","value":"Table 8 reports standard deviations over three replicates. The paper distinguishes this run variation from the choice of molecular dataset and split.","status":"source_checked","source_ids":["evidence-discovery-final-atom3d"],"source_locator":"Sections 3.1–3.8, 4–5; Appendix D–F; Table 8"},{"label":"Entity type","value":"Molecular dataset and evaluation suite.","status":"source_checked","source_ids":["src-discovery-drorlab-atom3d"],"source_locator":"Pinned README: Overview; dataset access; supported formats and splitting/filtering utilities"},{"label":"Organisms","value":"No single organism defines this suite of molecular structure datasets.","status":"inapplicable","source_ids":["src-discovery-drorlab-atom3d"],"source_locator":"Pinned README: Overview; dataset access; supported formats and splitting/filtering utilities"},{"label":"Assays","value":"Task-specific structural and molecular-property labels.","status":"source_checked","source_ids":["src-discovery-drorlab-atom3d"],"source_locator":"Pinned README: Overview; dataset access; supported formats and splitting/filtering utilities"},{"label":"Allowed inputs","value":"Three-dimensional molecular coordinates and task labels; supported formats include PDB, SDF and XYZ.","status":"source_checked","source_ids":["src-discovery-drorlab-atom3d"],"source_locator":"Pinned README: Overview; dataset access; supported formats and splitting/filtering utilities"},{"label":"Adaptation","value":"Supervised task evaluation; task-specific training configurations are linked separately.","status":"source_checked","source_ids":["src-discovery-drorlab-atom3d"],"source_locator":"Pinned README: Overview; dataset access; supported formats and splitting/filtering utilities"}],"strengths":[{"text":"Common data loaders allow different 3D representations to use the same dataset definitions.","source_ids":["src-discovery-drorlab-atom3d"],"source_locator":"Pinned README: Overview; dataset access; supported formats and splitting/filtering utilities"}],"limitations":[{"text":"Comparisons must retain the task, split and available structural information. In particular, the two LBA identity thresholds test different generalization regimes.","source_ids":["evidence-discovery-final-atom3d"],"source_locator":"Sections 3.1–3.8, 4–5; Appendix D–F; Table 8"}],"diagram":{"title":"Evaluation procedure","steps":["Allowed inputs: Three-dimensional molecular coordinates and task labels; supported formats include PDB, SDF and XYZ.","Splits: Task-specific: random molecules for SMP; 30% protein sequence identity for PIP, MSP and the strict LBA split; CATH topology groups for RES; protein targets for LEP; competition years for PSR and RSR. LBA also provides a less restrictive 60% identity split.","Metrics: SMP: MAE; PIP/MSP/LEP: classification AUROC; RES: accuracy; LBA: RMSE and correlations; PSR/RSR: correlations of predicted structure quality with reference GDT_TS/RMSD. Metrics and aggregation belong to each task."],"caption":"Conceptual procedure. Task variants and protocol versions retain their separate scoring conditions.","source_ids":["src-discovery-drorlab-atom3d","evidence-discovery-final-atom3d"],"source_locator":"Pinned README: Overview; dataset access; supported formats and splitting/filtering utilities; Sections 3.1–3.8, 4–5; Appendix D–F; Table 8"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary paper and/or task implementation reviewed for the explicitly cited methodology claims. Scope-limited absence is recorded only after the documented source search; no model runs or independent reproduction."}}},"description":"Three-dimensional molecular learning tasks","facets":{"areas":["molecular-interactions"]},"id":"discovery-benchmark-atom3d","kind":"benchmark","links":[],"name":"ATOM3D","source_ids":["src-discovery-drorlab-atom3d"],"status":"discovered"} {"attributes":{"entity_level":"suite","scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"RNA structure, function and engineering tasks","version":null,"historical_missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"BEACON compares RNA representations across structural and functional downstream tasks.","summary_source_ids":["src-discovery-terry-r123-rnabenchmark"],"summary_source_locator":"Pinned README: Dataset; task list; Models and Model settings","sections":[{"title":"Evaluation methodology","body":"BEACON tests RNA structure, function and engineering with 13 separately labelled datasets. Models predict nucleotide-level labels, pairwise structural maps or sequence-level properties. Its supplied task partitions and metrics must be preserved, and results are reported over three training seeds.","source_ids":["evidence-discovery-final-beacon"],"source_locator":"Sections 3.1–3.3, 4 and 5.1; Table 1; Appendix A"}],"facts":[{"label":"Datasets","value":"Tasks include secondary structure, contact/distance maps, RNA-family classification, modification and expression-related outcomes.","status":"source_checked","source_ids":["src-discovery-terry-r123-rnabenchmark"],"source_locator":"Pinned README: Dataset; task list; Models and Model settings"},{"label":"Splits","value":"Table 1 publishes separate training, validation and test sizes for all 13 tasks; these reuse different source datasets and are not one shared RNA partition. Structure-map tasks share their 188/23/80 partition, while other tasks use their own supplied folds.","status":"source_checked","source_ids":["evidence-discovery-final-beacon"],"source_locator":"Sections 3.1–3.3, 4 and 5.1; Table 1; Appendix A"},{"label":"Metrics","value":"F1 for secondary structure; top-L precision for contacts; R² for distance maps, imputation, APA, ribosome loading and switches; top-k accuracy for splice sites; accuracy for ncRNA class; AUC for modification; MCRMSE for degradation; weighted Spearman correlation for CRISPR tasks.","status":"source_checked","source_ids":["evidence-discovery-final-beacon"],"source_locator":"Sections 3.1–3.3, 4 and 5.1; Table 1; Appendix A"},{"label":"Baselines","value":"RNA-FM, RNABERT, RNA-MSM, SpliceBERT, UTR-LM, UTRBERT and BEACON variants are listed.","status":"source_checked","source_ids":["src-discovery-terry-r123-rnabenchmark"],"source_locator":"Pinned README: Dataset; task list; Models and Model settings"},{"label":"Leakage controls","value":"The paper specifies source datasets and per-task partitions, but does not establish one suite-wide homology or RNA-family exclusion rule. A train/test size table alone does not demonstrate independence from model pretraining.","status":"unreported","source_ids":["evidence-discovery-final-beacon"],"source_locator":"Sections 3.1–3.3, 4 and 5.1; Table 1; Appendix A"},{"label":"Uncertainty","value":"Experiments are repeated with three random seeds; Section 5.1 reports their mean and sample standard deviation.","status":"source_checked","source_ids":["evidence-discovery-final-beacon"],"source_locator":"Sections 3.1–3.3, 4 and 5.1; Table 1; Appendix A"},{"label":"Entity type","value":"RNA model benchmark suite (BEACON).","status":"source_checked","source_ids":["src-discovery-terry-r123-rnabenchmark"],"source_locator":"Pinned README: Dataset; task list; Models and Model settings"},{"label":"Organisms","value":"Task dependent: human HEK293 icSHAPE data and human splice/UTR assays coexist with RNA structure collections and synthetic constructs. The programmable-switch dataset includes sequences from viral genomes and human transcription factors; the suite is not a single-organism assay.","status":"source_checked","source_ids":["evidence-discovery-final-beacon"],"source_locator":"Sections 3.1–3.3, 4 and 5.1; Table 1; Appendix A"},{"label":"Assays","value":"Secondary structure, contact/distance, family, modification and expression-related labels.","status":"source_checked","source_ids":["src-discovery-terry-r123-rnabenchmark"],"source_locator":"Pinned README: Dataset; task list; Models and Model settings"},{"label":"Allowed inputs","value":"RNA sequences; task-specific targets are supplied in the benchmark datasets.","status":"source_checked","source_ids":["src-discovery-terry-r123-rnabenchmark"],"source_locator":"Pinned README: Dataset; task list; Models and Model settings"},{"label":"Adaptation","value":"Task-specific supervised evaluation using the supplied training scripts/configurations.","status":"source_checked","source_ids":["src-discovery-terry-r123-rnabenchmark"],"source_locator":"Pinned README: Dataset; task list; Models and Model settings"}],"strengths":[{"text":"Multiple RNA endpoints test distinct representation properties rather than one aggregate biological claim.","source_ids":["src-discovery-terry-r123-rnabenchmark"],"source_locator":"Pinned README: Dataset; task list; Models and Model settings"}],"limitations":[{"text":"BEACON combines heterogeneous source assays. Its published task partitions do not establish a common pretraining-overlap audit or a single family-held-out generalization test.","source_ids":["evidence-discovery-final-beacon"],"source_locator":"Sections 3.1–3.3, 4 and 5.1; Table 1; Appendix A"}],"diagram":{"title":"Evaluation procedure","steps":["Allowed inputs: RNA sequences; task-specific targets are supplied in the benchmark datasets.","Splits: Table 1 publishes separate training, validation and test sizes for all 13 tasks; these reuse different source datasets and are not one shared RNA partition. Structure-map tasks share their 188/23/80 partition, while other tasks use their own supplied folds.","Metrics: F1 for secondary structure; top-L precision for contacts; R² for distance maps, imputation, APA, ribosome loading and switches; top-k accuracy for splice sites; accuracy for ncRNA class; AUC for modification; MCRMSE for degradation; weighted Spearman correlation for CRISPR tasks."],"caption":"Conceptual procedure. Task variants and protocol versions retain their separate scoring conditions.","source_ids":["src-discovery-terry-r123-rnabenchmark","evidence-discovery-final-beacon"],"source_locator":"Pinned README: Dataset; task list; Models and Model settings; Sections 3.1–3.3, 4 and 5.1; Table 1; Appendix A"},"coverage":"limited","gaps":["Leakage controls: The paper specifies source datasets and per-task partitions, but does not establish one suite-wide homology or RNA-family exclusion rule. A train/test size table alone does not demonstrate independence from model pretraining."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary paper and/or task implementation reviewed for the explicitly cited methodology claims. Scope-limited absence is recorded only after the documented source search; no model runs or independent reproduction."}}},"description":"RNA structure, function and engineering tasks","facets":{"areas":["rna"]},"id":"discovery-benchmark-beacon","kind":"benchmark","links":[],"name":"BEACON","source_ids":["src-discovery-terry-r123-rnabenchmark"],"status":"discovered"} {"attributes":{"entity_level":"suite","scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Gene regulatory network inference","version":null,"historical_missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"BEELINE compares inferred gene-regulatory edge rankings with reference networks.","summary_source_ids":["src-discovery-murali-group-beeline"],"summary_source_locator":"Pinned README: Overview; Running algorithms; Evaluate results metric table; Plot results","sections":[{"title":"Evaluation methodology","body":"BEELINE evaluates gene regulatory network inference from single-cell expression. It runs inference algorithms on synthetic, curated-model and experimental datasets, then compares ranked predicted edges with known or constructed reference networks. Pseudotime requirements and reference-network reliability are part of each evaluation.","source_ids":["evidence-discovery-final-beeline"],"source_locator":"Methods: algorithm execution, simulated/curated datasets and experimental datasets"}],"facts":[{"label":"Datasets","value":"Single-cell expression inputs and ground-truth regulatory networks, configured per dataset.","status":"source_checked","source_ids":["src-discovery-murali-group-beeline"],"source_locator":"Pinned README: Overview; Running algorithms; Evaluate results metric table; Plot results"},{"label":"Splits","value":"BEELINE infers a network from each expression dataset and scores predicted edges against a reference network. Simulated replicates and experimental contexts are separate benchmark cases; the benchmark is not a single supervised train/validation/test classification split.","status":"source_checked","source_ids":["evidence-discovery-final-beeline"],"source_locator":"Methods: algorithm execution, simulated/curated datasets and experimental datasets"},{"label":"Metrics","value":"AUROC, AUPRC, early-precision ratio, signed early precision, rank correlation and top-edge overlap; runtime/network diagnostics are separate outputs.","status":"source_checked","source_ids":["src-discovery-murali-group-beeline"],"source_locator":"Pinned README: Overview; Running algorithms; Evaluate results metric table; Plot results"},{"label":"Baselines","value":"Containerized inference methods share ranked-edge outputs for a common evaluator.","status":"source_checked","source_ids":["src-discovery-murali-group-beeline"],"source_locator":"Pinned README: Overview; Running algorithms; Evaluate results metric table; Plot results"},{"label":"Leakage controls","value":"Reference networks are used to score inferred edges, while inputs are expression and, for some methods, pseudotime. Algorithm assumptions, simulated ground truth and experimental reference construction are explicit; no single held-out-gene training protocol applies to all methods.","status":"source_checked","source_ids":["evidence-discovery-final-beeline"],"source_locator":"Methods: algorithm execution, simulated/curated datasets and experimental datasets"},{"label":"Uncertainty","value":"Plotting supports multiple-run distributions, but a particular repeated-run design is not fixed by the README.","status":"source_checked","source_ids":["src-discovery-murali-group-beeline"],"source_locator":"Pinned README: Overview; Running algorithms; Evaluate results metric table; Plot results"},{"label":"Entity type","value":"Gene-regulatory network inference evaluation pipeline.","status":"source_checked","source_ids":["src-discovery-murali-group-beeline"],"source_locator":"Pinned README: Overview; Running algorithms; Evaluate results metric table; Plot results"},{"label":"Organisms","value":"Human transcription-factor metadata is supported; the chosen single-cell dataset defines organism scope.","status":"source_checked","source_ids":["src-discovery-murali-group-beeline"],"source_locator":"Pinned README: Overview; Running algorithms; Evaluate results metric table; Plot results"},{"label":"Assays","value":"Single-cell expression with a reference regulatory network.","status":"source_checked","source_ids":["src-discovery-murali-group-beeline"],"source_locator":"Pinned README: Overview; Running algorithms; Evaluate results metric table; Plot results"},{"label":"Allowed inputs","value":"Expression matrix, gene metadata and ground-truth edges for evaluation.","status":"source_checked","source_ids":["src-discovery-murali-group-beeline"],"source_locator":"Pinned README: Overview; Running algorithms; Evaluate results metric table; Plot results"},{"label":"Adaptation","value":"Inference algorithms operate on the supplied expression data; labeled reference edges are used for assessment.","status":"source_checked","source_ids":["src-discovery-murali-group-beeline"],"source_locator":"Pinned README: Overview; Running algorithms; Evaluate results metric table; Plot results"}],"strengths":[{"text":"Ranked-edge evaluation exposes precision–recall behavior rather than treating a thresholded graph as ground truth.","source_ids":["src-discovery-murali-group-beeline"],"source_locator":"Pinned README: Overview; Running algorithms; Evaluate results metric table; Plot results"}],"limitations":[{"text":"An inferred association is not automatically a causal regulatory edge. Simulated ground truth, curated biological models and experimentally assembled references support different conclusions.","source_ids":["evidence-discovery-final-beeline"],"source_locator":"Methods: algorithm execution, simulated/curated datasets and experimental datasets"}],"diagram":{"title":"Evaluation procedure","steps":["Allowed inputs: Expression matrix, gene metadata and ground-truth edges for evaluation.","Splits: BEELINE infers a network from each expression dataset and scores predicted edges against a reference network. Simulated replicates and experimental contexts are separate benchmark cases; the benchmark is not a single supervised train/validation/test classification split.","Metrics: AUROC, AUPRC, early-precision ratio, signed early precision, rank correlation and top-edge overlap; runtime/network diagnostics are separate outputs."],"caption":"Conceptual procedure. Task variants and protocol versions retain their separate scoring conditions.","source_ids":["src-discovery-murali-group-beeline","evidence-discovery-final-beeline"],"source_locator":"Pinned README: Overview; Running algorithms; Evaluate results metric table; Plot results; Methods: algorithm execution, simulated/curated datasets and experimental datasets"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary paper and/or task implementation reviewed for the explicitly cited methodology claims. Scope-limited absence is recorded only after the documented source search; no model runs or independent reproduction."}}},"description":"Gene regulatory network inference","facets":{"areas":["biological-networks"]},"id":"discovery-benchmark-beeline","kind":"benchmark","links":[],"name":"BEELINE","source_ids":["src-discovery-murali-group-beeline"],"status":"discovered"} {"attributes":{"entity_level":"suite","scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"DNA representations on biological downstream tasks","version":null,"historical_missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"BEND evaluates DNA representations using task-specific genomic annotations and explicit split membership.","summary_source_ids":["src-discovery-frederikkemarin-bend"],"summary_source_locator":"Pinned README: Data format; dataset use; Citation Guidelines","sections":[{"title":"Evaluation methodology","body":"BEND asks whether DNA representations support realistic human-genome annotation across short and long contexts. Frozen embeddings feed a small CNN for supervised tasks, while variant-effect prediction compares reference and alternate embeddings without fitting a task predictor. Whole-chromosome or sequence-identity partitions and specialist baselines make the tested capability explicit.","source_ids":["evidence-discovery-final-bend"],"source_locator":"Sections 3–4, Table 1, Table 3, Appendix A.1 and A.6"}],"facts":[{"label":"Datasets","value":"Genomic-coordinate tables paired with labels, including HDF5 labels for complex outputs; reference genome supplies input sequences.","status":"source_checked","source_ids":["src-discovery-frederikkemarin-bend"],"source_locator":"Pinned README: Data format; dataset use; Citation Guidelines"},{"label":"Splits","value":"BED task files contain an explicit split column, and paired label files share the same row index.","status":"source_checked","source_ids":["src-discovery-frederikkemarin-bend"],"source_locator":"Pinned README: Data format; dataset use; Citation Guidelines"},{"label":"Metrics","value":"Gene finding uses multiclass MCC; enhancer annotation uses AUPRC; chromatin accessibility, histone modification and methylation use label-wise AUROC; variant-effect tasks use AUROC.","status":"source_checked","source_ids":["evidence-discovery-final-bend"],"source_locator":"Sections 3–4, Table 1, Table 3, Appendix A.1 and A.6"},{"label":"Baselines","value":"A one-hot two-layer CNN is matched to frozen-embedding probes. Task specialists include AUGUSTUS, Enformer, Basset and DeepSEA, and the study includes supervised ResNet and simple pretrained language-model controls.","status":"source_checked","source_ids":["evidence-discovery-final-bend"],"source_locator":"Sections 3–4, Table 1, Table 3, Appendix A.1 and A.6"},{"label":"Leakage controls","value":"Supervised tasks hold out chromosomes, except gene finding, whose cross-partition pairs share no more than 80% identity of the mature protein. Enhancer evaluation uses ten chromosome-based folds. Variant effects are zero-shot tests; these partitions do not by themselves remove unsupervised genome-pretraining exposure.","status":"source_checked","source_ids":["evidence-discovery-final-bend"],"source_locator":"Appendix A.1.1 gene-finding split: mature-protein identity; other supervised task and enhancer split descriptions in Appendix A.1"},{"label":"Uncertainty","value":"Enhancer evaluation uses ten-fold cross-validation because the dataset is small. Table 3 does not provide a uniform repeated-seed confidence-interval protocol for all task/model scores; its Enformer ± entry must not be generalized to every row.","status":"unreported","source_ids":["evidence-discovery-final-bend"],"source_locator":"Sections 3–4, Table 1, Table 3, Appendix A.1 and A.6"},{"label":"Entity type","value":"Human genomic task benchmark suite.","status":"source_checked","source_ids":["src-discovery-frederikkemarin-bend"],"source_locator":"Pinned README: Data format; dataset use; Citation Guidelines"},{"label":"Organisms","value":"Human genomic evaluation; the README distinguishes models trained on other organisms from the evaluated set.","status":"source_checked","source_ids":["src-discovery-frederikkemarin-bend"],"source_locator":"Pinned README: Data format; dataset use; Citation Guidelines"},{"label":"Assays","value":"Task-dependent regulatory and annotation datasets, including ENCODE-derived resources.","status":"source_checked","source_ids":["src-discovery-frederikkemarin-bend"],"source_locator":"Pinned README: Data format; dataset use; Citation Guidelines"},{"label":"Allowed inputs","value":"DNA sequences extracted from genomic coordinates and a reference genome; complex labels align by index to HDF5 records.","status":"source_checked","source_ids":["src-discovery-frederikkemarin-bend"],"source_locator":"Pinned README: Data format; dataset use; Citation Guidelines"},{"label":"Adaptation","value":"Downstream supervised models use precomputed embeddings; unsupervised variant-effect scoring is a separate evaluation route.","status":"source_checked","source_ids":["src-discovery-frederikkemarin-bend"],"source_locator":"Pinned README: Data format; dataset use; Citation Guidelines"}],"strengths":[{"text":"Explicit split columns and aligned label indices make task membership inspectable.","source_ids":["src-discovery-frederikkemarin-bend"],"source_locator":"Pinned README: Data format; dataset use; Citation Guidelines"}],"limitations":[{"text":"BEND measures different biological tasks with different metrics. Its zero-shot variant tests, frozen probes and literature specialist results require separate interpretation; pretraining on the reference genome is not equivalent to supervised label leakage.","source_ids":["evidence-discovery-final-bend"],"source_locator":"Sections 3–4, Table 1, Table 3, Appendix A.1 and A.6"}],"diagram":{"title":"Evaluation procedure","steps":["Allowed inputs: DNA sequences extracted from genomic coordinates and a reference genome; complex labels align by index to HDF5 records.","Splits: BED task files contain an explicit split column, and paired label files share the same row index.","Metrics: Gene finding uses multiclass MCC; enhancer annotation uses AUPRC; chromatin accessibility, histone modification and methylation use label-wise AUROC; variant-effect tasks use AUROC."],"caption":"Conceptual procedure. Task variants and protocol versions retain their separate scoring conditions.","source_ids":["src-discovery-frederikkemarin-bend","evidence-discovery-final-bend"],"source_locator":"Pinned README: Data format; dataset use; Citation Guidelines; Sections 3–4, Table 1, Table 3, Appendix A.1 and A.6"},"coverage":"limited","gaps":["Uncertainty: Enhancer evaluation uses ten-fold cross-validation because the dataset is small. Table 3 does not provide a uniform repeated-seed confidence-interval protocol for all task/model scores; its Enformer ± entry must not be generalized to every row."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary paper and/or task implementation reviewed for the explicitly cited methodology claims. Scope-limited absence is recorded only after the documented source search; no model runs or independent reproduction. Independent automated spot review corrected interval terminology and sequence-identity scope against the original figure captions and methods."}}},"description":"DNA representations on biological downstream tasks","facets":{"areas":["genomics"]},"id":"discovery-benchmark-bend","kind":"benchmark","links":[],"name":"BEND","source_ids":["src-discovery-frederikkemarin-bend"],"status":"discovered"} {"attributes":{"entity_level":"challenge","scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Protein function prediction challenge","version":null,"historical_missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"CAFA evaluates prospective protein-function predictions against annotations that become available after prediction submission.","summary_source_ids":["evidence-benchmark-cafa-20260916"],"summary_source_locator":"Official CAFA page: The problem; The solution; challenge timeline","sections":[{"title":"Evaluation methodology","body":"CAFA is a time-delayed protein-function challenge: predictions are submitted before new experimental annotations become available. Evaluation then compares predictions with those newly added labels, separately by ontology and evaluation mode. CAFA3 provides explicit frequency and homology-transfer baselines and protein-level bootstrap intervals.","source_ids":["evidence-discovery-final-cafa3"],"source_locator":"CAFA3 Methods: Protein-centric evaluation; Figures 3–4"}],"facts":[{"label":"Datasets","value":"Protein sequences supplied for a challenge; proteins gaining experimental annotations after the deadline become evaluation targets.","status":"source_checked","source_ids":["evidence-benchmark-cafa-20260916"],"source_locator":"Official CAFA page: The problem; The solution; challenge timeline"},{"label":"Splits","value":"Prospective temporal assessment separates prediction submission from subsequent annotation growth.","status":"source_checked","source_ids":["evidence-benchmark-cafa-20260916"],"source_locator":"Official CAFA page: The problem; The solution; challenge timeline"},{"label":"Metrics","value":"CAFA3 protein-centric evaluation uses Fmax and semantic distance Smin, with prediction coverage reported. Term-centric assays also use AUROC. Ontology, knowledge category and full/partial evaluation mode must accompany the score.","status":"source_checked","source_ids":["evidence-discovery-final-cafa3"],"source_locator":"Methods: Protein-centric and term-centric evaluation; Figures 3–4"},{"label":"Baselines","value":"In CAFA3 protein-centric evaluation, Naïve predicts each term’s training-set frequency and BLAST transfers annotations using the highest matching sequence identity. Term-centric experiments also include expression-based comparators where available.","status":"source_checked","source_ids":["evidence-discovery-final-cafa3"],"source_locator":"CAFA3 Methods: Protein-centric evaluation; Figures 3–4"},{"label":"Leakage controls","value":"Newly acquired experimental annotations are used for later assessment; exact knowledge-cutoff controls depend on the challenge edition.","status":"source_checked","source_ids":["evidence-benchmark-cafa-20260916"],"source_locator":"Official CAFA page: The problem; The solution; challenge timeline"},{"label":"Uncertainty","value":"CAFA3 Figures 3–4 estimate 95% confidence intervals with 10,000 bootstrap samples of benchmark proteins. This is a version-specific evaluation, not a universal rule for every CAFA round.","status":"source_checked","source_ids":["evidence-discovery-final-cafa3"],"source_locator":"CAFA3 Methods: Protein-centric evaluation; Figures 3–4"},{"label":"Entity type","value":"Prospective protein-function prediction challenge.","status":"source_checked","source_ids":["evidence-benchmark-cafa-20260916"],"source_locator":"Official CAFA page: The problem; The solution; challenge timeline"},{"label":"Organisms","value":"Protein targets spanning challenge-selected taxa.","status":"source_checked","source_ids":["evidence-benchmark-cafa-20260916"],"source_locator":"Official CAFA page: The problem; The solution; challenge timeline"},{"label":"Assays","value":"Experimental functional annotations acquired after the prediction deadline.","status":"source_checked","source_ids":["evidence-benchmark-cafa-20260916"],"source_locator":"Official CAFA page: The problem; The solution; challenge timeline"},{"label":"Allowed inputs","value":"Challenge target protein sequences and permitted pre-deadline knowledge.","status":"source_checked","source_ids":["evidence-benchmark-cafa-20260916"],"source_locator":"Official CAFA page: The problem; The solution; challenge timeline"},{"label":"Adaptation","value":"Methods submit predictions before the target proteins gain evaluation annotations.","status":"source_checked","source_ids":["evidence-benchmark-cafa-20260916"],"source_locator":"Official CAFA page: The problem; The solution; challenge timeline"}],"strengths":[{"text":"Time-delayed annotation supplies a prospective assessment boundary.","source_ids":["evidence-benchmark-cafa-20260916"],"source_locator":"Official CAFA page: The problem; The solution; challenge timeline"}],"limitations":[{"text":"Annotation incompleteness, prediction coverage and the challenge’s ontology version affect interpretation. New CAFA rounds may revise the eligible proteins, metrics and uncertainty procedure.","source_ids":["evidence-discovery-final-cafa3"],"source_locator":"CAFA3 Methods: Protein-centric evaluation; Figures 3–4"}],"diagram":{"title":"Evaluation procedure","steps":["Allowed inputs: Challenge target protein sequences and permitted pre-deadline knowledge.","Splits: Prospective temporal assessment separates prediction submission from subsequent annotation growth.","Metrics: CAFA3 protein-centric evaluation uses Fmax and semantic distance Smin, with prediction coverage reported. Term-centric assays also use AUROC. Ontology, knowledge category and full/partial evaluation mode must accompany the score."],"caption":"Conceptual procedure. Task variants and protocol versions retain their separate scoring conditions.","source_ids":["evidence-benchmark-cafa-20260916","evidence-discovery-final-cafa3"],"source_locator":"Official CAFA page: The problem; The solution; challenge timeline; Methods: Protein-centric and term-centric evaluation; Figures 3–4"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary paper and/or task implementation reviewed for the explicitly cited methodology claims. Scope-limited absence is recorded only after the documented source search; no model runs or independent reproduction."}}},"description":"Protein function prediction challenge","facets":{"areas":["protein-function"]},"id":"discovery-benchmark-cafa","kind":"benchmark","links":[],"name":"CAFA","source_ids":["src-discovery-cafa"],"status":"discovered"} {"attributes":{"entity_level":"challenge","scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Metagenomic assembly, binning and profiling assessment","version":null,"historical_missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"CAMI is a community benchmark program for metagenomic computational methods.","summary_source_ids":["evidence-benchmark-cami-snapshot"],"summary_source_locator":"Official CAMI homepage: initiative description; challenges; dataset correction notices; toolkit citations","sections":[{"title":"Evaluation methodology","body":"CAMI is a series of blinded metagenomic software challenges. In CAMI II, participants received simulated short and long reads from defined communities and submitted assemblies, genome bins, taxonomic assignments or abundance profiles. Reference truth was used only for scoring, and software versions, input read types and community conditions were kept distinct.","source_ids":["evidence-discovery-final-cami2"],"source_locator":"Methods: Challenge datasets, Challenge organization, Evaluation metrics; Figures 2–4; Table 1"}],"facts":[{"label":"Datasets","value":"Challenge-specific metagenomic datasets, including a longitudinal human-gut collection in CAMI III.","status":"source_checked","source_ids":["evidence-benchmark-cami-snapshot"],"source_locator":"Official CAMI homepage: initiative description; challenges; dataset correction notices; toolkit citations"},{"label":"Splits","value":"CAMI II supplied public-genome practice datasets with ground truth before its blinded challenge. Challenge datasets were marine, strain-madness and plant-associated communities; these are challenge conditions, not a standard supervised train/validation/test partition.","status":"source_checked","source_ids":["evidence-discovery-final-cami2"],"source_locator":"Methods: Challenge datasets, Challenge organization, Evaluation metrics; Figures 2–4; Table 1"},{"label":"Metrics","value":"Assembly: genome fraction, NGA50, mismatches, misassemblies and strain precision/recall. Genome binning: purity, completeness, ARI and binned fraction. Taxonomic binning: purity, completeness, F1 and accuracy. Profiling: identification, abundance and diversity metrics, including L1, Bray–Curtis and weighted UniFrac.","status":"source_checked","source_ids":["evidence-discovery-final-cami2"],"source_locator":"Methods: Evaluation metrics; Table 1"},{"label":"Baselines","value":"Submitted programs are compared under the same data condition. Gold-standard assemblies and MEGAHIT assemblies separate binning performance from upstream assembly error; published method identities and versions are listed in Table 1.","status":"source_checked","source_ids":["evidence-discovery-final-cami2"],"source_locator":"Methods: Challenge datasets, Challenge organization, Evaluation metrics; Figures 2–4; Table 1"},{"label":"Leakage controls","value":"Challenge genome data and metadata were kept confidential until the challenge ended. Public reference collections dated 8 January 2019 were supplied for reference-based methods. CAMI II also includes public genomes, so novelty is stratified rather than assumed for every organism.","status":"source_checked","source_ids":["evidence-discovery-final-cami2"],"source_locator":"Methods: Challenge datasets, Challenge organization, Evaluation metrics; Figures 2–4; Table 1"},{"label":"Uncertainty","value":"Uncertainty is task-specific: taxonomic binning Figure 3 uses standard errors across bins; taxonomic profiling Figure 4 reports means across samples with standard deviations. These are not a common seed-based interval for every CAMI metric.","status":"source_checked","source_ids":["evidence-discovery-final-cami2"],"source_locator":"Figure 3 and Figure 4 captions: standard error across taxonomic bins versus standard deviation across samples"},{"label":"Entity type","value":"Metagenomic community challenge series.","status":"source_checked","source_ids":["evidence-benchmark-cami-snapshot"],"source_locator":"Official CAMI homepage: initiative description; challenges; dataset correction notices; toolkit citations"},{"label":"Organisms","value":"Microbial communities; CAMI III includes longitudinal human-gut samples.","status":"source_checked","source_ids":["evidence-benchmark-cami-snapshot"],"source_locator":"Official CAMI homepage: initiative description; challenges; dataset correction notices; toolkit citations"},{"label":"Assays","value":"Challenge-specific metagenomic sequence data and reference composition.","status":"source_checked","source_ids":["evidence-benchmark-cami-snapshot"],"source_locator":"Official CAMI homepage: initiative description; challenges; dataset correction notices; toolkit citations"},{"label":"Allowed inputs","value":"Released sequence data and track-specific reference resources.","status":"source_checked","source_ids":["evidence-benchmark-cami-snapshot"],"source_locator":"Official CAMI homepage: initiative description; challenges; dataset correction notices; toolkit citations"},{"label":"Adaptation","value":"Methods process challenge inputs; a challenge edition and track determine resource rules.","status":"source_checked","source_ids":["evidence-benchmark-cami-snapshot"],"source_locator":"Official CAMI homepage: initiative description; challenges; dataset correction notices; toolkit citations"}],"strengths":[{"text":"Multiple tracks distinguish assembly, binning and abundance estimation.","source_ids":["evidence-benchmark-cami-snapshot"],"source_locator":"Official CAMI homepage: initiative description; challenges; dataset correction notices; toolkit citations"}],"limitations":[{"text":"This profile documents CAMI II as a concrete protocol example. Other CAMI rounds may use different genomes, reference databases and metrics; strain diversity and input assembly quality materially change the task.","source_ids":["evidence-discovery-final-cami2"],"source_locator":"Methods: Challenge datasets, Challenge organization, Evaluation metrics; Figures 2–4; Table 1"}],"diagram":{"title":"Evaluation procedure","steps":["Allowed inputs: Released sequence data and track-specific reference resources.","Splits: CAMI II supplied public-genome practice datasets with ground truth before its blinded challenge. Challenge datasets were marine, strain-madness and plant-associated communities; these are challenge conditions, not a standard supervised train/validation/test partition.","Metrics: Assembly: genome fraction, NGA50, mismatches, misassemblies and strain precision/recall. Genome binning: purity, completeness, ARI and binned fraction. Taxonomic binning: purity, completeness, F1 and accuracy. Profiling: identification, abundance and diversity metrics, including L1, Bray–Curtis and weighted UniFrac."],"caption":"Conceptual procedure. Task variants and protocol versions retain their separate scoring conditions.","source_ids":["evidence-benchmark-cami-snapshot","evidence-discovery-final-cami2"],"source_locator":"Official CAMI homepage: initiative description; challenges; dataset correction notices; toolkit citations; Methods: Challenge datasets, Challenge organization, Evaluation metrics; Figures 2–4; Table 1; Methods: Evaluation metrics; Table 1"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary paper and/or task implementation reviewed for the explicitly cited methodology claims. Scope-limited absence is recorded only after the documented source search; no model runs or independent reproduction. Independent automated spot review corrected interval terminology and sequence-identity scope against the original figure captions and methods."}}},"description":"Metagenomic assembly, binning and profiling assessment","facets":{"areas":["microbiome"]},"id":"discovery-benchmark-cami","kind":"benchmark","links":[],"name":"CAMI","source_ids":["src-discovery-cami"],"status":"discovered"} {"attributes":{"entity_level":"task","scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"genome binning","version":null,"historical_missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"The CAMI genome-binning task can be understood through its documented assessment tool; this guide does not identify a challenge-specific run.","summary_source_ids":["src-discovery-cami-challenge-amber"],"summary_source_locator":"Pinned README: introduction; Metrics computed per bin/per sample; input formats","sections":[{"title":"Evaluation methodology","body":"CAMI is a series of blinded metagenomic software challenges. In CAMI II, participants received simulated short and long reads from defined communities and submitted assemblies, genome bins, taxonomic assignments or abundance profiles. Reference truth was used only for scoring, and software versions, input read types and community conditions were kept distinct.","source_ids":["evidence-discovery-final-cami2"],"source_locator":"Methods: Challenge datasets, Challenge organization, Evaluation metrics; Figures 2–4; Table 1"}],"facts":[{"label":"Datasets","value":"A selected CAMI challenge dataset and matching gold standard are required; this entry does not fix the edition.","status":"source_checked","source_ids":["src-discovery-cami-challenge-amber"],"source_locator":"Pinned README: introduction; Metrics computed per bin/per sample; input formats"},{"label":"Splits","value":"CAMI II supplied public-genome practice datasets with ground truth before its blinded challenge. Challenge datasets were marine, strain-madness and plant-associated communities; these are challenge conditions, not a standard supervised train/validation/test partition.","status":"source_checked","source_ids":["evidence-discovery-final-cami2"],"source_locator":"Methods: Challenge datasets, Challenge organization, Evaluation metrics; Figures 2–4; Table 1"},{"label":"Metrics","value":"Bin purity/completeness; sample accuracy, contamination, adjusted Rand index, binned fraction and recovered-genome counts; UniFrac for taxonomic binning.","status":"source_checked","source_ids":["src-discovery-cami-challenge-amber"],"source_locator":"Pinned README: introduction; Metrics computed per bin/per sample; input formats"},{"label":"Baselines","value":"Submitted programs are compared under the same data condition. Gold-standard assemblies and MEGAHIT assemblies separate binning performance from upstream assembly error; published method identities and versions are listed in Table 1.","status":"source_checked","source_ids":["evidence-discovery-final-cami2"],"source_locator":"Methods: Challenge datasets, Challenge organization, Evaluation metrics; Figures 2–4; Table 1"},{"label":"Leakage controls","value":"Challenge genome data and metadata were kept confidential until the challenge ended. Public reference collections dated 8 January 2019 were supplied for reference-based methods. CAMI II also includes public genomes, so novelty is stratified rather than assumed for every organism.","status":"source_checked","source_ids":["evidence-discovery-final-cami2"],"source_locator":"Methods: Challenge datasets, Challenge organization, Evaluation metrics; Figures 2–4; Table 1"},{"label":"Uncertainty","value":"Uncertainty is task-specific: taxonomic binning Figure 3 uses standard errors across bins; taxonomic profiling Figure 4 reports means across samples with standard deviations. These are not a common seed-based interval for every CAMI metric.","status":"source_checked","source_ids":["evidence-discovery-final-cami2"],"source_locator":"Figure 3 and Figure 4 captions: standard error across taxonomic bins versus standard deviation across samples"},{"label":"Entity type","value":"Constituent benchmark task: CAMI genome binning","status":"source_checked","source_ids":["src-discovery-cami-challenge-amber"],"source_locator":"Pinned README: introduction; Metrics computed per bin/per sample; input formats"},{"label":"Organisms","value":"AMBER accepts community gold standards; the selected CAMI dataset supplies organism membership.","status":"inapplicable","source_ids":["src-discovery-cami-challenge-amber"],"source_locator":"Pinned README: introduction; Metrics computed per bin/per sample; input formats"},{"label":"Assays","value":"Metagenomic sequence assignments and their gold-standard bins/taxa.","status":"source_checked","source_ids":["src-discovery-cami-challenge-amber"],"source_locator":"Pinned README: introduction; Metrics computed per bin/per sample; input formats"},{"label":"Allowed inputs","value":"Predicted sequence-to-bin or sequence-to-taxon assignments and the matched gold standard.","status":"source_checked","source_ids":["src-discovery-cami-challenge-amber"],"source_locator":"Pinned README: introduction; Metrics computed per bin/per sample; input formats"},{"label":"Adaptation","value":"AMBER assesses assignments; predictor-training conditions are outside the evaluator.","status":"inapplicable","source_ids":["src-discovery-cami-challenge-amber"],"source_locator":"Pinned README: introduction; Metrics computed per bin/per sample; input formats"}],"strengths":[{"text":"Purity and completeness expose different binning errors.","source_ids":["src-discovery-cami-challenge-amber"],"source_locator":"Pinned README: introduction; Metrics computed per bin/per sample; input formats"}],"limitations":[{"text":"This profile documents CAMI II as a concrete protocol example. Other CAMI rounds may use different genomes, reference databases and metrics; strain diversity and input assembly quality materially change the task.","source_ids":["evidence-discovery-final-cami2"],"source_locator":"Methods: Challenge datasets, Challenge organization, Evaluation metrics; Figures 2–4; Table 1"}],"diagram":{"title":"Evaluation procedure","steps":["Allowed inputs: Predicted sequence-to-bin or sequence-to-taxon assignments and the matched gold standard.","Splits: CAMI II supplied public-genome practice datasets with ground truth before its blinded challenge. Challenge datasets were marine, strain-madness and plant-associated communities; these are challenge conditions, not a standard supervised train/validation/test partition.","Metrics: Bin purity/completeness; sample accuracy, contamination, adjusted Rand index, binned fraction and recovered-genome counts; UniFrac for taxonomic binning."],"caption":"Conceptual procedure. Task variants and protocol versions retain their separate scoring conditions.","source_ids":["src-discovery-cami-challenge-amber","evidence-discovery-final-cami2"],"source_locator":"Pinned README: introduction; Metrics computed per bin/per sample; input formats; Methods: Challenge datasets, Challenge organization, Evaluation metrics; Figures 2–4; Table 1"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary paper and/or task implementation reviewed for the explicitly cited methodology claims. Scope-limited absence is recorded only after the documented source search; no model runs or independent reproduction. Independent automated spot review corrected interval terminology and sequence-identity scope against the original figure captions and methods."}}},"description":"genome binning","facets":{"areas":["microbiome"]},"id":"discovery-benchmark-cami-genome-binning","kind":"benchmark","links":[{"relation":"parent","target_id":"discovery-benchmark-cami"},{"relation":"part_of","target_id":"discovery-benchmark-cami"}],"name":"CAMI genome binning","source_ids":["src-discovery-cami"],"status":"discovered"} {"attributes":{"entity_level":"task","scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"metagenome assembly","version":null,"historical_missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"The CAMI assembly task concerns reconstructing metagenomic sequence assemblies; the homepage links MetaQUAST evaluation resources.","summary_source_ids":["evidence-benchmark-cami-snapshot"],"summary_source_locator":"Official CAMI homepage: initiative description; challenges; dataset correction notices; toolkit citations","sections":[{"title":"Evaluation methodology","body":"CAMI is a series of blinded metagenomic software challenges. In CAMI II, participants received simulated short and long reads from defined communities and submitted assemblies, genome bins, taxonomic assignments or abundance profiles. Reference truth was used only for scoring, and software versions, input read types and community conditions were kept distinct.","source_ids":["evidence-discovery-final-cami2"],"source_locator":"Methods: Challenge datasets, Challenge organization, Evaluation metrics; Figures 2–4; Table 1"}],"facts":[{"label":"Datasets","value":"CAMI II marine, strain-madness and plant-associated simulated metagenomes provide short-read, long-read and hybrid conditions. Genome truth and gold-standard assemblies are available after the blinded challenge; preserve dataset and sequencing condition.","status":"source_checked","source_ids":["evidence-discovery-final-cami2"],"source_locator":"Methods: Challenge datasets and Challenge organization"},{"label":"Splits","value":"CAMI II supplied public-genome practice datasets with ground truth before its blinded challenge. Challenge datasets were marine, strain-madness and plant-associated communities; these are challenge conditions, not a standard supervised train/validation/test partition.","status":"source_checked","source_ids":["evidence-discovery-final-cami2"],"source_locator":"Methods: Challenge datasets, Challenge organization, Evaluation metrics; Figures 2–4; Table 1"},{"label":"Metrics","value":"MetaQUAST evaluates genome fraction, mismatches per 100 kb, duplication ratio, NGA50 and misassemblies; strain precision and recall additionally measure high-quality strain reconstruction. Undefined per-genome NGA50 is set to zero before the reported genome average.","status":"source_checked","source_ids":["evidence-discovery-final-cami2"],"source_locator":"Methods: Assembly metrics; Figure 1"},{"label":"Baselines","value":"Submitted programs are compared under the same data condition. Gold-standard assemblies and MEGAHIT assemblies separate binning performance from upstream assembly error; published method identities and versions are listed in Table 1.","status":"source_checked","source_ids":["evidence-discovery-final-cami2"],"source_locator":"Methods: Challenge datasets, Challenge organization, Evaluation metrics; Figures 2–4; Table 1"},{"label":"Leakage controls","value":"Challenge genome data and metadata were kept confidential until the challenge ended. Public reference collections dated 8 January 2019 were supplied for reference-based methods. CAMI II also includes public genomes, so novelty is stratified rather than assumed for every organism.","status":"source_checked","source_ids":["evidence-discovery-final-cami2"],"source_locator":"Methods: Challenge datasets, Challenge organization, Evaluation metrics; Figures 2–4; Table 1"},{"label":"Uncertainty","value":"Figure 1 and the assembly-metrics Methods define descriptive per-genome and dataset summaries, but no universal bootstrap or repeated-seed interval for all assembly scores.","status":"unreported","source_ids":["evidence-discovery-final-cami2"],"source_locator":"Methods: Assembly metrics; Figure 1"},{"label":"Entity type","value":"Constituent benchmark task: CAMI metagenome assembly","status":"source_checked","source_ids":["evidence-benchmark-cami-snapshot"],"source_locator":"Official CAMI homepage: initiative description; challenges; dataset correction notices; toolkit citations"},{"label":"Organisms","value":"Microbial communities; CAMI III includes longitudinal human-gut samples.","status":"source_checked","source_ids":["evidence-benchmark-cami-snapshot"],"source_locator":"Official CAMI homepage: initiative description; challenges; dataset correction notices; toolkit citations"},{"label":"Assays","value":"Challenge-specific metagenomic sequence data and reference composition.","status":"source_checked","source_ids":["evidence-benchmark-cami-snapshot"],"source_locator":"Official CAMI homepage: initiative description; challenges; dataset correction notices; toolkit citations"},{"label":"Allowed inputs","value":"Released sequence data and track-specific reference resources.","status":"source_checked","source_ids":["evidence-benchmark-cami-snapshot"],"source_locator":"Official CAMI homepage: initiative description; challenges; dataset correction notices; toolkit citations"},{"label":"Adaptation","value":"Methods process challenge inputs; a challenge edition and track determine resource rules.","status":"source_checked","source_ids":["evidence-benchmark-cami-snapshot"],"source_locator":"Official CAMI homepage: initiative description; challenges; dataset correction notices; toolkit citations"}],"strengths":[{"text":"Multiple tracks distinguish assembly, binning and abundance estimation.","source_ids":["evidence-benchmark-cami-snapshot"],"source_locator":"Official CAMI homepage: initiative description; challenges; dataset correction notices; toolkit citations"}],"limitations":[{"text":"This profile documents CAMI II as a concrete protocol example. Other CAMI rounds may use different genomes, reference databases and metrics; strain diversity and input assembly quality materially change the task.","source_ids":["evidence-discovery-final-cami2"],"source_locator":"Methods: Challenge datasets, Challenge organization, Evaluation metrics; Figures 2–4; Table 1"}],"diagram":{"title":"Evaluation procedure","steps":["Allowed inputs: Released sequence data and track-specific reference resources.","Splits: CAMI II supplied public-genome practice datasets with ground truth before its blinded challenge. Challenge datasets were marine, strain-madness and plant-associated communities; these are challenge conditions, not a standard supervised train/validation/test partition.","Metrics: MetaQUAST evaluates genome fraction, mismatches per 100 kb, duplication ratio, NGA50 and misassemblies; strain precision and recall additionally measure high-quality strain reconstruction. Undefined per-genome NGA50 is set to zero before the reported genome average."],"caption":"Conceptual procedure. Task variants and protocol versions retain their separate scoring conditions.","source_ids":["evidence-benchmark-cami-snapshot","evidence-discovery-final-cami2"],"source_locator":"Official CAMI homepage: initiative description; challenges; dataset correction notices; toolkit citations; Methods: Challenge datasets, Challenge organization, Evaluation metrics; Figures 2–4; Table 1; Methods: Assembly metrics; Figure 1"},"coverage":"limited","gaps":["Uncertainty: Figure 1 and the assembly-metrics Methods define descriptive per-genome and dataset summaries, but no universal bootstrap or repeated-seed interval for all assembly scores."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary paper and/or task implementation reviewed for the explicitly cited methodology claims. Scope-limited absence is recorded only after the documented source search; no model runs or independent reproduction."}}},"description":"metagenome assembly","facets":{"areas":["microbiome"]},"id":"discovery-benchmark-cami-metagenome-assembly","kind":"benchmark","links":[{"relation":"parent","target_id":"discovery-benchmark-cami"},{"relation":"part_of","target_id":"discovery-benchmark-cami"}],"name":"CAMI metagenome assembly","source_ids":["src-discovery-cami"],"status":"discovered"} {"attributes":{"entity_level":"task","scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"taxonomic binning","version":null,"historical_missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"The CAMI taxonomic-binning task can be understood through its documented assessment tool; this guide does not identify a challenge-specific run.","summary_source_ids":["src-discovery-cami-challenge-amber"],"summary_source_locator":"Pinned README: introduction; Metrics computed per bin/per sample; input formats","sections":[{"title":"Evaluation methodology","body":"CAMI is a series of blinded metagenomic software challenges. In CAMI II, participants received simulated short and long reads from defined communities and submitted assemblies, genome bins, taxonomic assignments or abundance profiles. Reference truth was used only for scoring, and software versions, input read types and community conditions were kept distinct.","source_ids":["evidence-discovery-final-cami2"],"source_locator":"Methods: Challenge datasets, Challenge organization, Evaluation metrics; Figures 2–4; Table 1"}],"facts":[{"label":"Datasets","value":"A selected CAMI challenge dataset and matching gold standard are required; this entry does not fix the edition.","status":"source_checked","source_ids":["src-discovery-cami-challenge-amber"],"source_locator":"Pinned README: introduction; Metrics computed per bin/per sample; input formats"},{"label":"Splits","value":"CAMI II supplied public-genome practice datasets with ground truth before its blinded challenge. Challenge datasets were marine, strain-madness and plant-associated communities; these are challenge conditions, not a standard supervised train/validation/test partition.","status":"source_checked","source_ids":["evidence-discovery-final-cami2"],"source_locator":"Methods: Challenge datasets, Challenge organization, Evaluation metrics; Figures 2–4; Table 1"},{"label":"Metrics","value":"Bin purity/completeness; sample accuracy, contamination, adjusted Rand index, binned fraction and recovered-genome counts; UniFrac for taxonomic binning.","status":"source_checked","source_ids":["src-discovery-cami-challenge-amber"],"source_locator":"Pinned README: introduction; Metrics computed per bin/per sample; input formats"},{"label":"Baselines","value":"Submitted programs are compared under the same data condition. Gold-standard assemblies and MEGAHIT assemblies separate binning performance from upstream assembly error; published method identities and versions are listed in Table 1.","status":"source_checked","source_ids":["evidence-discovery-final-cami2"],"source_locator":"Methods: Challenge datasets, Challenge organization, Evaluation metrics; Figures 2–4; Table 1"},{"label":"Leakage controls","value":"Challenge genome data and metadata were kept confidential until the challenge ended. Public reference collections dated 8 January 2019 were supplied for reference-based methods. CAMI II also includes public genomes, so novelty is stratified rather than assumed for every organism.","status":"source_checked","source_ids":["evidence-discovery-final-cami2"],"source_locator":"Methods: Challenge datasets, Challenge organization, Evaluation metrics; Figures 2–4; Table 1"},{"label":"Uncertainty","value":"Uncertainty is task-specific: taxonomic binning Figure 3 uses standard errors across bins; taxonomic profiling Figure 4 reports means across samples with standard deviations. These are not a common seed-based interval for every CAMI metric.","status":"source_checked","source_ids":["evidence-discovery-final-cami2"],"source_locator":"Figure 3 and Figure 4 captions: standard error across taxonomic bins versus standard deviation across samples"},{"label":"Entity type","value":"Constituent benchmark task: CAMI taxonomic binning","status":"source_checked","source_ids":["src-discovery-cami-challenge-amber"],"source_locator":"Pinned README: introduction; Metrics computed per bin/per sample; input formats"},{"label":"Organisms","value":"AMBER accepts community gold standards; the selected CAMI dataset supplies organism membership.","status":"inapplicable","source_ids":["src-discovery-cami-challenge-amber"],"source_locator":"Pinned README: introduction; Metrics computed per bin/per sample; input formats"},{"label":"Assays","value":"Metagenomic sequence assignments and their gold-standard bins/taxa.","status":"source_checked","source_ids":["src-discovery-cami-challenge-amber"],"source_locator":"Pinned README: introduction; Metrics computed per bin/per sample; input formats"},{"label":"Allowed inputs","value":"Predicted sequence-to-bin or sequence-to-taxon assignments and the matched gold standard.","status":"source_checked","source_ids":["src-discovery-cami-challenge-amber"],"source_locator":"Pinned README: introduction; Metrics computed per bin/per sample; input formats"},{"label":"Adaptation","value":"AMBER assesses assignments; predictor-training conditions are outside the evaluator.","status":"inapplicable","source_ids":["src-discovery-cami-challenge-amber"],"source_locator":"Pinned README: introduction; Metrics computed per bin/per sample; input formats"}],"strengths":[{"text":"Purity and completeness expose different binning errors.","source_ids":["src-discovery-cami-challenge-amber"],"source_locator":"Pinned README: introduction; Metrics computed per bin/per sample; input formats"}],"limitations":[{"text":"This profile documents CAMI II as a concrete protocol example. Other CAMI rounds may use different genomes, reference databases and metrics; strain diversity and input assembly quality materially change the task.","source_ids":["evidence-discovery-final-cami2"],"source_locator":"Methods: Challenge datasets, Challenge organization, Evaluation metrics; Figures 2–4; Table 1"}],"diagram":{"title":"Evaluation procedure","steps":["Allowed inputs: Predicted sequence-to-bin or sequence-to-taxon assignments and the matched gold standard.","Splits: CAMI II supplied public-genome practice datasets with ground truth before its blinded challenge. Challenge datasets were marine, strain-madness and plant-associated communities; these are challenge conditions, not a standard supervised train/validation/test partition.","Metrics: Bin purity/completeness; sample accuracy, contamination, adjusted Rand index, binned fraction and recovered-genome counts; UniFrac for taxonomic binning."],"caption":"Conceptual procedure. Task variants and protocol versions retain their separate scoring conditions.","source_ids":["src-discovery-cami-challenge-amber","evidence-discovery-final-cami2"],"source_locator":"Pinned README: introduction; Metrics computed per bin/per sample; input formats; Methods: Challenge datasets, Challenge organization, Evaluation metrics; Figures 2–4; Table 1"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary paper and/or task implementation reviewed for the explicitly cited methodology claims. Scope-limited absence is recorded only after the documented source search; no model runs or independent reproduction. Independent automated spot review corrected interval terminology and sequence-identity scope against the original figure captions and methods."}}},"description":"taxonomic binning","facets":{"areas":["microbiome"]},"id":"discovery-benchmark-cami-taxonomic-binning","kind":"benchmark","links":[{"relation":"parent","target_id":"discovery-benchmark-cami"},{"relation":"part_of","target_id":"discovery-benchmark-cami"}],"name":"CAMI taxonomic binning","source_ids":["src-discovery-cami"],"status":"discovered"} {"attributes":{"entity_level":"task","scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"taxonomic profiling","version":null,"historical_missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"The CAMI taxonomic-profiling task can be understood through its documented assessment tool; this guide does not identify a challenge-specific run.","summary_source_ids":["src-discovery-cami-challenge-opal"],"summary_source_locator":"Pinned README: introduction; Computed metrics; input format; example evaluations","sections":[{"title":"Evaluation methodology","body":"CAMI is a series of blinded metagenomic software challenges. In CAMI II, participants received simulated short and long reads from defined communities and submitted assemblies, genome bins, taxonomic assignments or abundance profiles. Reference truth was used only for scoring, and software versions, input read types and community conditions were kept distinct.","source_ids":["evidence-discovery-final-cami2"],"source_locator":"Methods: Challenge datasets, Challenge organization, Evaluation metrics; Figures 2–4; Table 1"}],"facts":[{"label":"Datasets","value":"A selected CAMI challenge dataset and matching gold standard are required; this entry does not fix the edition.","status":"source_checked","source_ids":["src-discovery-cami-challenge-opal"],"source_locator":"Pinned README: introduction; Computed metrics; input format; example evaluations"},{"label":"Splits","value":"CAMI II supplied public-genome practice datasets with ground truth before its blinded challenge. Challenge datasets were marine, strain-madness and plant-associated communities; these are challenge conditions, not a standard supervised train/validation/test partition.","status":"source_checked","source_ids":["evidence-discovery-final-cami2"],"source_locator":"Methods: Challenge datasets, Challenge organization, Evaluation metrics; Figures 2–4; Table 1"},{"label":"Metrics","value":"Precision, recall, F1, Jaccard, L1 error, UniFrac, Bray–Curtis and diversity measures.","status":"source_checked","source_ids":["src-discovery-cami-challenge-opal"],"source_locator":"Pinned README: introduction; Computed metrics; input format; example evaluations"},{"label":"Baselines","value":"Submitted programs are compared under the same data condition. Gold-standard assemblies and MEGAHIT assemblies separate binning performance from upstream assembly error; published method identities and versions are listed in Table 1.","status":"source_checked","source_ids":["evidence-discovery-final-cami2"],"source_locator":"Methods: Challenge datasets, Challenge organization, Evaluation metrics; Figures 2–4; Table 1"},{"label":"Leakage controls","value":"Challenge genome data and metadata were kept confidential until the challenge ended. Public reference collections dated 8 January 2019 were supplied for reference-based methods. CAMI II also includes public genomes, so novelty is stratified rather than assumed for every organism.","status":"source_checked","source_ids":["evidence-discovery-final-cami2"],"source_locator":"Methods: Challenge datasets, Challenge organization, Evaluation metrics; Figures 2–4; Table 1"},{"label":"Uncertainty","value":"Uncertainty is task-specific: taxonomic binning Figure 3 uses standard errors across bins; taxonomic profiling Figure 4 reports means across samples with standard deviations. These are not a common seed-based interval for every CAMI metric.","status":"source_checked","source_ids":["evidence-discovery-final-cami2"],"source_locator":"Figure 3 and Figure 4 captions: standard error across taxonomic bins versus standard deviation across samples"},{"label":"Entity type","value":"Constituent benchmark task: CAMI taxonomic profiling","status":"source_checked","source_ids":["src-discovery-cami-challenge-opal"],"source_locator":"Pinned README: introduction; Computed metrics; input format; example evaluations"},{"label":"Organisms","value":"Taxa are supplied by the selected reference community.","status":"inapplicable","source_ids":["src-discovery-cami-challenge-opal"],"source_locator":"Pinned README: introduction; Computed metrics; input format; example evaluations"},{"label":"Assays","value":"Challenge-specific metagenomic sequence data and reference composition.","status":"source_checked","source_ids":["src-discovery-cami-challenge-opal"],"source_locator":"Pinned README: introduction; Computed metrics; input format; example evaluations"},{"label":"Allowed inputs","value":"Predicted and reference abundance profiles.","status":"source_checked","source_ids":["src-discovery-cami-challenge-opal"],"source_locator":"Pinned README: introduction; Computed metrics; input format; example evaluations"},{"label":"Adaptation","value":"OPAL scores profiles; it does not train the submitting method.","status":"inapplicable","source_ids":["src-discovery-cami-challenge-opal"],"source_locator":"Pinned README: introduction; Computed metrics; input format; example evaluations"}],"strengths":[{"text":"Abundance agreement and taxon detection are separate evaluation dimensions.","source_ids":["src-discovery-cami-challenge-opal"],"source_locator":"Pinned README: introduction; Computed metrics; input format; example evaluations"}],"limitations":[{"text":"This profile documents CAMI II as a concrete protocol example. Other CAMI rounds may use different genomes, reference databases and metrics; strain diversity and input assembly quality materially change the task.","source_ids":["evidence-discovery-final-cami2"],"source_locator":"Methods: Challenge datasets, Challenge organization, Evaluation metrics; Figures 2–4; Table 1"}],"diagram":{"title":"Evaluation procedure","steps":["Allowed inputs: Predicted and reference abundance profiles.","Splits: CAMI II supplied public-genome practice datasets with ground truth before its blinded challenge. Challenge datasets were marine, strain-madness and plant-associated communities; these are challenge conditions, not a standard supervised train/validation/test partition.","Metrics: Precision, recall, F1, Jaccard, L1 error, UniFrac, Bray–Curtis and diversity measures."],"caption":"Conceptual procedure. Task variants and protocol versions retain their separate scoring conditions.","source_ids":["src-discovery-cami-challenge-opal","evidence-discovery-final-cami2"],"source_locator":"Pinned README: introduction; Computed metrics; input format; example evaluations; Methods: Challenge datasets, Challenge organization, Evaluation metrics; Figures 2–4; Table 1"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary paper and/or task implementation reviewed for the explicitly cited methodology claims. Scope-limited absence is recorded only after the documented source search; no model runs or independent reproduction. Independent automated spot review corrected interval terminology and sequence-identity scope against the original figure captions and methods."}}},"description":"taxonomic profiling","facets":{"areas":["microbiome"]},"id":"discovery-benchmark-cami-taxonomic-profiling","kind":"benchmark","links":[{"relation":"parent","target_id":"discovery-benchmark-cami"},{"relation":"part_of","target_id":"discovery-benchmark-cami"}],"name":"CAMI taxonomic profiling","source_ids":["src-discovery-cami"],"status":"discovered"} {"attributes":{"entity_level":"challenge","scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Protein interaction docking assessment","version":null,"historical_missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"CAPRI assesses blind predictions of protein-complex structures supplied before experimental publication.","summary_source_ids":["evidence-benchmark-capri-snapshot"],"summary_source_locator":"Official CAPRI homepage: introductory description","sections":[{"title":"Evaluation methodology","body":"CAPRI is a blinded interaction-structure challenge. Predictors submit complex models, while scoring tracks select promising models from candidate sets. Assessors compare submitted interfaces with withheld experimental structures; quality classes and DockQ-based summaries retain the target and round context.","source_ids":["evidence-discovery-final-capri"],"source_locator":"Assessment page: joint CASP-CAPRI Round 57 and Round 58"}],"facts":[{"label":"Datasets","value":"Challenge-round protein-complex targets supplied by experimental contributors.","status":"source_checked","source_ids":["evidence-benchmark-capri-snapshot"],"source_locator":"Official CAPRI homepage: introductory description"},{"label":"Splits","value":"Target structures are unreleased to predictors during the blind prediction phase.","status":"source_checked","source_ids":["evidence-benchmark-capri-snapshot"],"source_locator":"Official CAPRI homepage: introductory description"},{"label":"Metrics","value":"Interface contact recovery (Fnat), interface RMSD and ligand RMSD define classical CAPRI quality classes. DockQ combines these complementary structural measures into a continuous score.","status":"source_checked","source_ids":["evidence-discovery-final-dockq"],"source_locator":"Introduction and Methods: CAPRI criteria and DockQ"},{"label":"Baselines","value":"Round 57 assessment compares submissions with ColabFold and an AlphaFold 3 submission. Round 58 supplies MassiveFold candidate models and tests improvement over that model pool; other rounds have different reference conditions.","status":"source_checked","source_ids":["evidence-discovery-final-capri"],"source_locator":"Assessment page: joint CASP-CAPRI Round 57 and Round 58"},{"label":"Leakage controls","value":"The prospective blind-target design separates submissions from public experimental structures.","status":"source_checked","source_ids":["evidence-benchmark-capri-snapshot"],"source_locator":"Official CAPRI homepage: introductory description"},{"label":"Uncertainty","value":"The inspected assessment page provides round-specific rankings and quality classes but does not define one common bootstrap interval for all CAPRI targets and rounds.","status":"unreported","source_ids":["evidence-discovery-final-capri"],"source_locator":"Assessment page: joint CASP-CAPRI Round 57 and Round 58"},{"label":"Entity type","value":"Blind protein-complex prediction challenge.","status":"source_checked","source_ids":["evidence-benchmark-capri-snapshot"],"source_locator":"Official CAPRI homepage: introductory description"},{"label":"Organisms","value":"Target-dependent protein complexes; this is not a single-species benchmark.","status":"source_checked","source_ids":["evidence-benchmark-capri-snapshot"],"source_locator":"Official CAPRI homepage: introductory description"},{"label":"Assays","value":"Experimentally determined complex structures withheld for assessment.","status":"source_checked","source_ids":["evidence-benchmark-capri-snapshot"],"source_locator":"Official CAPRI homepage: introductory description"},{"label":"Allowed inputs","value":"Target information distributed for each challenge round.","status":"source_checked","source_ids":["evidence-benchmark-capri-snapshot"],"source_locator":"Official CAPRI homepage: introductory description"},{"label":"Adaptation","value":"Prediction submissions are assessed against subsequently available experimental structures.","status":"source_checked","source_ids":["evidence-benchmark-capri-snapshot"],"source_locator":"Official CAPRI homepage: introductory description"}],"strengths":[{"text":"Blind rounds separate predictor submissions from the assessment structures.","source_ids":["evidence-benchmark-capri-snapshot"],"source_locator":"Official CAPRI homepage: introductory description"}],"limitations":[{"text":"A model-selection round with supplied structures is different from docking from sequence or component structures. Reference inputs, candidate pools and assessment-unit definitions must stay visible.","source_ids":["evidence-discovery-final-capri"],"source_locator":"Assessment page: joint CASP-CAPRI Round 57 and Round 58"}],"diagram":{"title":"Evaluation procedure","steps":["Allowed inputs: Target information distributed for each challenge round.","Splits: Target structures are unreleased to predictors during the blind prediction phase.","Metrics: Interface contact recovery (Fnat), interface RMSD and ligand RMSD define classical CAPRI quality classes. DockQ combines these complementary structural measures into a continuous score."],"caption":"Conceptual procedure. Task variants and protocol versions retain their separate scoring conditions.","source_ids":["evidence-benchmark-capri-snapshot","evidence-discovery-final-dockq"],"source_locator":"Official CAPRI homepage: introductory description; Introduction and Methods: CAPRI criteria and DockQ"},"coverage":"limited","gaps":["Uncertainty: The inspected assessment page provides round-specific rankings and quality classes but does not define one common bootstrap interval for all CAPRI targets and rounds."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary paper and/or task implementation reviewed for the explicitly cited methodology claims. Scope-limited absence is recorded only after the documented source search; no model runs or independent reproduction."}}},"description":"Protein interaction docking assessment","facets":{"areas":["protein-structure"]},"id":"discovery-benchmark-capri","kind":"benchmark","links":[],"name":"CAPRI","source_ids":["src-discovery-capri"],"status":"discovered"} {"attributes":{"entity_level":"challenge","scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Community protein structure assessment","version":null,"historical_missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"CASP assesses structure-prediction methods through blind predictions and category-specific evaluation.","summary_source_ids":["evidence-benchmark-casp-snapshot"],"summary_source_locator":"Protein Structure Prediction Center homepage: Welcome; assessment and archived-data description","sections":[{"title":"Evaluation methodology","body":"CASP evaluates predictions against experimentally determined structures withheld at submission time. Its independent assessors define evaluation units and category-specific scores. In CASP16 monomer assessment, controlled MSA inputs and automated ColabFold references help distinguish pipeline contributions from differences in information supplied.","source_ids":["evidence-discovery-final-casp16"],"source_locator":"CASP16 monomer assessment: Evaluation of Model 6, model sampling and head-to-head comparisons; Methods"}],"facts":[{"label":"Datasets","value":"Experiment-specific targets, submissions and numerical assessment files are archived by the Prediction Center.","status":"source_checked","source_ids":["evidence-benchmark-casp-snapshot"],"source_locator":"Protein Structure Prediction Center homepage: Welcome; assessment and archived-data description"},{"label":"Splits","value":"Blind prediction tasks are organized by assessment edition and category.","status":"source_checked","source_ids":["evidence-benchmark-casp-snapshot"],"source_locator":"Protein Structure Prediction Center homepage: Welcome; assessment and archived-data description"},{"label":"Metrics","value":"CASP16 monomer assessment combines standardized GDT_HA, QSE, reLLG_const, SphGr, CAD_AA, GDC_SC, AL0_P, lDDT and MolProbity measures. These scores apply to its defined evaluation units; other CASP categories use different protocols.","status":"source_checked","source_ids":["evidence-discovery-final-casp16"],"source_locator":"Monomer assessment: scoring formula and Methods"},{"label":"Baselines","value":"CASP16 monomer assessment uses ColabFold as an automated reference, includes MassiveFold sampling, and compares models constrained to the same ColabFold MSAs to separate input improvements from prediction-network changes.","status":"source_checked","source_ids":["evidence-discovery-final-casp16"],"source_locator":"CASP16 monomer assessment: Evaluation of Model 6, model sampling and head-to-head comparisons; Methods"},{"label":"Leakage controls","value":"The blind-target setting is explicit; exact training-cutoff enforcement and template restrictions depend on the category.","status":"source_checked","source_ids":["evidence-benchmark-casp-snapshot"],"source_locator":"Protein Structure Prediction Center homepage: Welcome; assessment and archived-data description"},{"label":"Uncertainty","value":"CASP16 monomer head-to-head comparisons use 1,000 bootstrap samples of evaluation units. This protocol is category- and round-specific; it is not a confidence interval for every CASP result.","status":"source_checked","source_ids":["evidence-discovery-final-casp16"],"source_locator":"CASP16 monomer assessment: Evaluation of Model 6, model sampling and head-to-head comparisons; Methods"},{"label":"Entity type","value":"Community protein-structure prediction experiment.","status":"source_checked","source_ids":["evidence-benchmark-casp-snapshot"],"source_locator":"Protein Structure Prediction Center homepage: Welcome; assessment and archived-data description"},{"label":"Organisms","value":"Target-dependent proteins and complexes.","status":"source_checked","source_ids":["evidence-benchmark-casp-snapshot"],"source_locator":"Protein Structure Prediction Center homepage: Welcome; assessment and archived-data description"},{"label":"Assays","value":"Experiment-specific experimentally determined structures.","status":"source_checked","source_ids":["evidence-benchmark-casp-snapshot"],"source_locator":"Protein Structure Prediction Center homepage: Welcome; assessment and archived-data description"},{"label":"Allowed inputs","value":"Sequence and target information issued for the selected experiment.","status":"source_checked","source_ids":["evidence-benchmark-casp-snapshot"],"source_locator":"Protein Structure Prediction Center homepage: Welcome; assessment and archived-data description"},{"label":"Adaptation","value":"Prediction and assessment phases are separated; rules vary by target category and experiment.","status":"source_checked","source_ids":["evidence-benchmark-casp-snapshot"],"source_locator":"Protein Structure Prediction Center homepage: Welcome; assessment and archived-data description"}],"strengths":[{"text":"Archived targets, submissions and assessment files support round-specific comparisons.","source_ids":["evidence-benchmark-casp-snapshot"],"source_locator":"Protein Structure Prediction Center homepage: Welcome; assessment and archived-data description"}],"limitations":[{"text":"CASP rounds and categories differ in targets, allowed information and scoring. Domain-level, whole-complex and nucleic-acid results cannot be pooled as interchangeable observations.","source_ids":["evidence-discovery-final-casp16"],"source_locator":"CASP16 monomer assessment: Evaluation of Model 6, model sampling and head-to-head comparisons; Methods"}],"diagram":{"title":"Evaluation procedure","steps":["Allowed inputs: Sequence and target information issued for the selected experiment.","Splits: Blind prediction tasks are organized by assessment edition and category.","Metrics: CASP16 monomer assessment combines standardized GDT_HA, QSE, reLLG_const, SphGr, CAD_AA, GDC_SC, AL0_P, lDDT and MolProbity measures. These scores apply to its defined evaluation units; other CASP categories use different protocols."],"caption":"Conceptual procedure. Task variants and protocol versions retain their separate scoring conditions.","source_ids":["evidence-benchmark-casp-snapshot","evidence-discovery-final-casp16"],"source_locator":"Protein Structure Prediction Center homepage: Welcome; assessment and archived-data description; Monomer assessment: scoring formula and Methods"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary paper and/or task implementation reviewed for the explicitly cited methodology claims. Scope-limited absence is recorded only after the documented source search; no model runs or independent reproduction."}}},"description":"Community protein structure assessment","facets":{"areas":["protein-structure"]},"id":"discovery-benchmark-casp","kind":"benchmark","links":[],"name":"CASP","source_ids":["src-discovery-casp"],"status":"discovered"} {"attributes":{"entity_level":"suite","scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Human regulatory DNA representation evaluation","version":null,"historical_missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"DART-Eval measures human regulatory-DNA representations under zero-shot, probing and fine-tuning regimes.","summary_source_ids":["src-discovery-kundajelab-dart-eval"],"summary_source_locator":"Pinned README: Overview; Data download; task-organized outputs; probing/fine-tuning sections","sections":[{"title":"Evaluation methodology","body":"DART-Eval separates regulatory sequence discrimination, motif sensitivity, cell-type specificity, quantitative accessibility and variant effects. It uses matched controls and consistent chromosome partitions where models are fitted. A result therefore needs its precise task and adaptation setting, rather than a single regulatory-intelligence score.","source_ids":["evidence-discovery-final-dart"],"source_locator":"Appendix: training/test splits, clustering and supervised evaluation; reproducibility checklist"}],"facts":[{"label":"Datasets","value":"Task-specific HDF5 inputs/outputs with raw data and evaluated model outputs organized in a Synapse project.","status":"source_checked","source_ids":["src-discovery-kundajelab-dart-eval"],"source_locator":"Pinned README: Overview; Data download; task-organized outputs; probing/fine-tuning sections"},{"label":"Splits","value":"Training uses chromosomes other than the held-out sets; validation is chromosomes 6 and 21; test is chromosomes 5, 10, 14, 18, 20 and 22. Fitted checkpoints are chosen using validation loss.","status":"source_checked","source_ids":["evidence-discovery-final-dart"],"source_locator":"Appendix: common train/validation/test split"},{"label":"Metrics","value":"Regulatory and variant classification use AUROC/AUPRC; quantitative accessibility uses Pearson and Spearman correlations on peaks alone and peaks plus background. Cell-type clustering uses adjusted mutual information. Motif sensitivity is a separate paired-sequence evaluation.","status":"source_checked","source_ids":["evidence-discovery-final-dart"],"source_locator":"Appendix: training/test splits, clustering and supervised evaluation; reproducibility checklist"},{"label":"Baselines","value":"Probing-head-like ab initio models are documented alongside pretrained-model probing and fine-tuning.","status":"source_checked","source_ids":["src-discovery-kundajelab-dart-eval"],"source_locator":"Pinned README: Overview; Data download; task-organized outputs; probing/fine-tuning sections"},{"label":"Leakage controls","value":"The common genomic partition holds out chromosomes 5, 10, 14, 18, 20 and 22 for test, with chromosomes 6 and 21 for validation. Lowest validation loss selects checkpoints. Dinucleotide-shuffled negatives preserve composition; this control does not eliminate all pretrained-genome exposure.","status":"source_checked","source_ids":["evidence-discovery-final-dart"],"source_locator":"Appendix: training/test splits, clustering and supervised evaluation; reproducibility checklist"},{"label":"Uncertainty","value":"Clustering is repeated 100 times and reports a 95% interval across clustering runs. Motif-sensitivity intervals are provided in the linked artifacts; the paper does not report repeated-training intervals for every other task.","status":"source_checked","source_ids":["evidence-discovery-final-dart"],"source_locator":"Appendix: training/test splits, clustering and supervised evaluation; reproducibility checklist"},{"label":"Entity type","value":"DNA regulatory evaluation suite.","status":"source_checked","source_ids":["src-discovery-kundajelab-dart-eval"],"source_locator":"Pinned README: Overview; Data download; task-organized outputs; probing/fine-tuning sections"},{"label":"Organisms","value":"Human regulatory datasets.","status":"source_checked","source_ids":["src-discovery-kundajelab-dart-eval"],"source_locator":"Pinned README: Overview; Data download; task-organized outputs; probing/fine-tuning sections"},{"label":"Assays","value":"Regulatory-element, chromatin-accessibility and variant-effect measurements.","status":"source_checked","source_ids":["src-discovery-kundajelab-dart-eval"],"source_locator":"Pinned README: Overview; Data download; task-organized outputs; probing/fine-tuning sections"},{"label":"Allowed inputs","value":"Task-specific DNA sequences and HDF5 targets; raw data and model outputs are linked through Synapse.","status":"source_checked","source_ids":["src-discovery-kundajelab-dart-eval"],"source_locator":"Pinned README: Overview; Data download; task-organized outputs; probing/fine-tuning sections"},{"label":"Adaptation","value":"Zero-shot, probing, fine-tuning and ab-initio comparison regimes are distinct.","status":"source_checked","source_ids":["src-discovery-kundajelab-dart-eval"],"source_locator":"Pinned README: Overview; Data download; task-organized outputs; probing/fine-tuning sections"}],"strengths":[{"text":"Tasks separate generic regulatory recognition from cell-specific activity and variant effects.","source_ids":["src-discovery-kundajelab-dart-eval"],"source_locator":"Pinned README: Overview; Data download; task-organized outputs; probing/fine-tuning sections"}],"limitations":[{"text":"The assessed tasks use local regulatory contexts. A chromosome-held-out supervised test does not establish that those sequences were absent from unsupervised pretraining.","source_ids":["evidence-discovery-final-dart"],"source_locator":"Appendix: training/test splits, clustering and supervised evaluation; reproducibility checklist"}],"diagram":{"title":"Evaluation procedure","steps":["Allowed inputs: Task-specific DNA sequences and HDF5 targets; raw data and model outputs are linked through Synapse.","Splits: Training uses chromosomes other than the held-out sets; validation is chromosomes 6 and 21; test is chromosomes 5, 10, 14, 18, 20 and 22. Fitted checkpoints are chosen using validation loss.","Metrics: Regulatory and variant classification use AUROC/AUPRC; quantitative accessibility uses Pearson and Spearman correlations on peaks alone and peaks plus background. Cell-type clustering uses adjusted mutual information. Motif sensitivity is a separate paired-sequence evaluation."],"caption":"Conceptual procedure. Task variants and protocol versions retain their separate scoring conditions.","source_ids":["src-discovery-kundajelab-dart-eval","evidence-discovery-final-dart"],"source_locator":"Pinned README: Overview; Data download; task-organized outputs; probing/fine-tuning sections; Appendix: common train/validation/test split; Appendix: training/test splits, clustering and supervised evaluation; reproducibility checklist"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary paper and/or task implementation reviewed for the explicitly cited methodology claims. Scope-limited absence is recorded only after the documented source search; no model runs or independent reproduction."}}},"description":"Human regulatory DNA representation evaluation","facets":{"areas":["genomics"]},"id":"discovery-benchmark-dart-eval","kind":"benchmark","links":[],"name":"DART-Eval","source_ids":["src-discovery-kundajelab-dart-eval"],"status":"discovered"} {"attributes":{"entity_level":"suite","scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Generalisation in protein fitness landscapes","version":null,"historical_missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"FLIP evaluates protein-sequence representations using multiple deliberately defined train/test splits.","summary_source_ids":["src-discovery-j-snackkb-flip"],"summary_source_locator":"Pinned README: repository organization; Splits","sections":[{"title":"Evaluation methodology","body":"FLIP evaluates sequence-to-fitness models under protein-engineering distribution shifts. It separates landscapes from their partitions: the same assay can test random interpolation, higher mutation counts, higher fitness or transfer across sequence groups. Comparisons are meaningful within the same landscape and split.","source_ids":["evidence-discovery-final-flip"],"source_locator":"Sections 3–5; Tables 2 and 4–7"}],"facts":[{"label":"Datasets","value":"Protein sequence/property collections distributed as raw data, processed splits and FASTA resources.","status":"source_checked","source_ids":["src-discovery-j-snackkb-flip"],"source_locator":"Pinned README: repository organization; Splits"},{"label":"Splits","value":"The splits directory documents biological/statistical split logic; multiple splits may exist for one dataset.","status":"source_checked","source_ids":["src-discovery-j-snackkb-flip"],"source_locator":"Pinned README: repository organization; Splits"},{"label":"Metrics","value":"Spearman rank correlation between predicted and measured fitness on each landscape/split.","status":"source_checked","source_ids":["evidence-discovery-final-flip"],"source_locator":"Sections 3–5; Tables 2 and 4–7"},{"label":"Baselines","value":"A baselines directory provides reference implementations.","status":"source_checked","source_ids":["src-discovery-j-snackkb-flip"],"source_locator":"Pinned README: repository organization; Splits"},{"label":"Leakage controls","value":"FLIP contrasts random partitions with mutation-number, fitness and sequence-family or diversity partitions. Its purpose is to expose distribution shifts; a random partition is an easier control, not an equivalent test.","status":"source_checked","source_ids":["evidence-discovery-final-flip"],"source_locator":"Sections 3–5; Tables 2 and 4–7"},{"label":"Uncertainty","value":"The inspected baseline tables report point correlations. Their Methods do not define a common repeated-seed or bootstrap confidence interval for all landscape/split results.","status":"unreported","source_ids":["evidence-discovery-final-flip"],"source_locator":"Sections 3–5; Tables 2 and 4–7"},{"label":"Entity type","value":"Protein sequence learning benchmark with multiple split regimes.","status":"source_checked","source_ids":["src-discovery-j-snackkb-flip"],"source_locator":"Pinned README: repository organization; Splits"},{"label":"Organisms","value":"GB1 binding variants, adeno-associated virus capsid variants and Meltome proteins across the tree of life; the landscapes have different organism and assay scopes.","status":"source_checked","source_ids":["evidence-discovery-final-flip"],"source_locator":"Sections 3–5; Tables 2 and 4–7"},{"label":"Assays","value":"Measured protein properties from the selected source datasets.","status":"source_checked","source_ids":["src-discovery-j-snackkb-flip"],"source_locator":"Pinned README: repository organization; Splits"},{"label":"Allowed inputs","value":"Protein sequences with property labels.","status":"source_checked","source_ids":["src-discovery-j-snackkb-flip"],"source_locator":"Pinned README: repository organization; Splits"},{"label":"Adaptation","value":"Supervised learning on the chosen training partition; different splits test different generalization conditions.","status":"source_checked","source_ids":["src-discovery-j-snackkb-flip"],"source_locator":"Pinned README: repository organization; Splits"}],"strengths":[{"text":"Split semaphores explicitly mark active, cautionary and obsolete comparison settings.","source_ids":["src-discovery-j-snackkb-flip"],"source_locator":"Pinned README: repository organization; Splits"}],"limitations":[{"text":"Random and extrapolative splits answer different questions. Closely related mutants are intentional in engineering tasks, so independence cannot be reduced to a universal sequence-identity threshold.","source_ids":["evidence-discovery-final-flip"],"source_locator":"Sections 3–5; Tables 2 and 4–7"}],"diagram":{"title":"Evaluation procedure","steps":["Allowed inputs: Protein sequences with property labels.","Splits: The splits directory documents biological/statistical split logic; multiple splits may exist for one dataset.","Metrics: Spearman rank correlation between predicted and measured fitness on each landscape/split."],"caption":"Conceptual procedure. Task variants and protocol versions retain their separate scoring conditions.","source_ids":["src-discovery-j-snackkb-flip","evidence-discovery-final-flip"],"source_locator":"Pinned README: repository organization; Splits; Sections 3–5; Tables 2 and 4–7"},"coverage":"limited","gaps":["Uncertainty: The inspected baseline tables report point correlations. Their Methods do not define a common repeated-seed or bootstrap confidence interval for all landscape/split results."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary paper and/or task implementation reviewed for the explicitly cited methodology claims. Scope-limited absence is recorded only after the documented source search; no model runs or independent reproduction."}}},"description":"Generalisation in protein fitness landscapes","facets":{"areas":["protein-function"]},"id":"discovery-benchmark-flip","kind":"benchmark","links":[],"name":"FLIP","source_ids":["src-discovery-j-snackkb-flip"],"status":"discovered"} {"attributes":{"entity_level":"suite","scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Expanded protein fitness landscapes","version":null,"historical_missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"FLIP2 evaluates protein fitness prediction under deliberately shifted training and test distributions.","summary_source_ids":["evidence-benchmark-flip2-snapshot"],"summary_source_locator":"FLIP2 official website: Overview; benchmark features; datasets and split categories","sections":[{"title":"Evaluation methodology","body":"FLIP2 extends protein fitness testing to enzyme activity, spectral properties, hydrophobic-core effects and protein interactions. Its partitions hold out mutation positions, mutation numbers, fitness ranges or parent sequences. Simple one-hot ridge models and pretrained versus randomly initialized networks make the baseline comparison interpretable.","source_ids":["evidence-discovery-final-flip2"],"source_locator":"Sections 2–4; Table 1; supplementary result tables"}],"facts":[{"label":"Datasets","value":"Datasets span enzymatic activity, protein interactions and other measured protein properties.","status":"source_checked","source_ids":["evidence-benchmark-flip2-snapshot"],"source_locator":"FLIP2 official website: Overview; benchmark features; datasets and split categories"},{"label":"Splits","value":"Split categories separate mutation counts, positions, mutation identities, fitness ranges or reference proteins.","status":"source_checked","source_ids":["evidence-benchmark-flip2-snapshot"],"source_locator":"FLIP2 official website: Overview; benchmark features; datasets and split categories"},{"label":"Metrics","value":"Spearman rank correlation is the primary metric, with NDCG additionally measuring prioritization of high-fitness variants.","status":"source_checked","source_ids":["evidence-discovery-final-flip2"],"source_locator":"Sections 2–4; Table 1; supplementary result tables"},{"label":"Baselines","value":"Zero-shot protein models, ridge regression and fine-tuned models.","status":"source_checked","source_ids":["evidence-benchmark-flip2-snapshot"],"source_locator":"FLIP2 official website: Overview; benchmark features; datasets and split categories"},{"label":"Leakage controls","value":"Distribution-shift splits are explicit; exact homology/pretraining overlap depends on the dataset.","status":"source_checked","source_ids":["evidence-benchmark-flip2-snapshot"],"source_locator":"FLIP2 official website: Overview; benchmark features; datasets and split categories"},{"label":"Uncertainty","value":"Fine-tuned language models are run five times with different random seeds and validation-based early stopping; reported metrics are averaged. The zero-shot and ridge evaluations follow separate procedures.","status":"source_checked","source_ids":["evidence-discovery-final-flip2"],"source_locator":"Sections 2–4; Table 1; supplementary result tables"},{"label":"Entity type","value":"Protein fitness benchmark collection.","status":"source_checked","source_ids":["evidence-benchmark-flip2-snapshot"],"source_locator":"FLIP2 official website: Overview; benchmark features; datasets and split categories"},{"label":"Organisms","value":"Mixed natural and designed protein landscapes: amylase, imine reductase, NucB, TrpB, hydrophobic-core variants, microbial rhodopsins and a PDZ3–peptide interaction system. These are not a single-species benchmark.","status":"source_checked","source_ids":["evidence-discovery-final-flip2"],"source_locator":"Sections 2–4; Table 1; supplementary result tables"},{"label":"Assays","value":"Enzymatic activity, protein interactions and other measured properties.","status":"source_checked","source_ids":["evidence-benchmark-flip2-snapshot"],"source_locator":"FLIP2 official website: Overview; benchmark features; datasets and split categories"},{"label":"Allowed inputs","value":"Protein sequence representations and dataset-specific labels.","status":"source_checked","source_ids":["evidence-benchmark-flip2-snapshot"],"source_locator":"FLIP2 official website: Overview; benchmark features; datasets and split categories"},{"label":"Adaptation","value":"Supervised prediction under task-specific data splits.","status":"source_checked","source_ids":["evidence-benchmark-flip2-snapshot"],"source_locator":"FLIP2 official website: Overview; benchmark features; datasets and split categories"}],"strengths":[{"text":"Explicit task diversity avoids treating fitness as a single interchangeable measurement.","source_ids":["evidence-benchmark-flip2-snapshot"],"source_locator":"FLIP2 official website: Overview; benchmark features; datasets and split categories"}],"limitations":[{"text":"The same number of training examples does not make random and engineering-oriented splits equivalent. Per-landscape ranking and top-fitness retrieval should be retained alongside any aggregate.","source_ids":["evidence-discovery-final-flip2"],"source_locator":"Sections 2–4; Table 1; supplementary result tables"}],"diagram":{"title":"Evaluation procedure","steps":["Allowed inputs: Protein sequence representations and dataset-specific labels.","Splits: Split categories separate mutation counts, positions, mutation identities, fitness ranges or reference proteins.","Metrics: Spearman rank correlation is the primary metric, with NDCG additionally measuring prioritization of high-fitness variants."],"caption":"Conceptual procedure. Task variants and protocol versions retain their separate scoring conditions.","source_ids":["evidence-benchmark-flip2-snapshot","evidence-discovery-final-flip2"],"source_locator":"FLIP2 official website: Overview; benchmark features; datasets and split categories; Sections 2–4; Table 1; supplementary result tables"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary paper and/or task implementation reviewed for the explicitly cited methodology claims. Scope-limited absence is recorded only after the documented source search; no model runs or independent reproduction."}}},"description":"Expanded protein fitness landscapes","facets":{"areas":["protein-function"]},"id":"discovery-benchmark-flip2","kind":"benchmark","links":[],"name":"FLIP2","source_ids":["src-discovery-flip2"],"status":"discovered"} {"attributes":{"entity_level":"suite","scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Frozen genomic representations with linear probes","version":null,"historical_missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"GENEB compares frozen genomic representations across classification tasks and label-budget regimes.","summary_source_ids":["src-discovery-darlednik-geneb"],"summary_source_locator":"Pinned README: Overview; data regimes; leaderboard aggregation definition","sections":[{"title":"Evaluation methodology","body":"GENEB compares frozen genomic model representations under standardized probe and sample-budget settings. It keeps preprocessing, partitions and random seeds consistent across models and summarizes several biological task categories. Organism and dataset coverage are uneven, so category-specific results remain necessary.","source_ids":["evidence-discovery-final-geneb"],"source_locator":"Evaluation protocol; benchmark construction appendix; Limitations"}],"facts":[{"label":"Datasets","value":"DNA classification tasks grouped into functional categories such as promoters, enhancers, methylation and splice sites.","status":"source_checked","source_ids":["src-discovery-darlednik-geneb"],"source_locator":"Pinned README: Overview; data regimes; leaderboard aggregation definition"},{"label":"Splits","value":"Full-data, ten-shot and one-shot settings use a common frozen-embedding protocol.","status":"source_checked","source_ids":["src-discovery-darlednik-geneb"],"source_locator":"Pinned README: Overview; data regimes; leaderboard aggregation definition"},{"label":"Metrics","value":"MCC; macro aggregation averages category scores, while the labelled micro aggregate averages task scores.","status":"source_checked","source_ids":["src-discovery-darlednik-geneb"],"source_locator":"Pinned README: Overview; data regimes; leaderboard aggregation definition"},{"label":"Baselines","value":"A broad set of genomic foundation models evaluated with the same representation protocol.","status":"source_checked","source_ids":["src-discovery-darlednik-geneb"],"source_locator":"Pinned README: Overview; data regimes; leaderboard aggregation definition"},{"label":"Leakage controls","value":"The protocol fixes preprocessing, partitions and seeds across models, but the inspected paper does not establish a benchmark-wide homology exclusion or pretraining-contamination audit. Shared evaluation settings are not proof that training corpora exclude test sequences.","status":"unreported","source_ids":["evidence-discovery-final-geneb"],"source_locator":"Evaluation protocol; benchmark construction appendix; Limitations"},{"label":"Uncertainty","value":"The stated protocol averages over five fixed random seeds for probing in the 1-shot, 10-shot and full-data regimes. Averaging seeds is not itself a confidence interval.","status":"source_checked","source_ids":["evidence-discovery-final-geneb"],"source_locator":"Evaluation protocol; benchmark construction appendix; Limitations"},{"label":"Entity type","value":"DNA-model classification benchmark suite.","status":"source_checked","source_ids":["src-discovery-darlednik-geneb"],"source_locator":"Pinned README: Overview; data regimes; leaderboard aggregation definition"},{"label":"Organisms","value":"The collection combines tasks from human and other well-studied organisms, including mouse and plant categories. Its limitations explicitly note this organism bias; individual dataset provenance remains the appropriate species definition.","status":"source_checked","source_ids":["evidence-discovery-final-geneb"],"source_locator":"Evaluation protocol; benchmark construction appendix; Limitations"},{"label":"Assays","value":"Promoter, enhancer, methylation and splice-site classification labels.","status":"source_checked","source_ids":["src-discovery-darlednik-geneb"],"source_locator":"Pinned README: Overview; data regimes; leaderboard aggregation definition"},{"label":"Allowed inputs","value":"DNA sequences and classification labels.","status":"source_checked","source_ids":["src-discovery-darlednik-geneb"],"source_locator":"Pinned README: Overview; data regimes; leaderboard aggregation definition"},{"label":"Adaptation","value":"Benchmark-specific supervised evaluation of pretrained DNA models.","status":"source_checked","source_ids":["src-discovery-darlednik-geneb"],"source_locator":"Pinned README: Overview; data regimes; leaderboard aggregation definition"}],"strengths":[{"text":"Task and functional-group aggregation distinguish data-rich groups from balanced coverage.","source_ids":["src-discovery-darlednik-geneb"],"source_locator":"Pinned README: Overview; data regimes; leaderboard aggregation definition"}],"limitations":[{"text":"The paper acknowledges concentration on human and well-studied organisms. A consistent probe protocol cannot establish an unseen-genome boundary for every pretrained model.","source_ids":["evidence-discovery-final-geneb"],"source_locator":"Evaluation protocol; benchmark construction appendix; Limitations"}],"diagram":{"title":"Evaluation procedure","steps":["Allowed inputs: DNA sequences and classification labels.","Splits: Full-data, ten-shot and one-shot settings use a common frozen-embedding protocol.","Metrics: MCC; macro aggregation averages category scores, while the labelled micro aggregate averages task scores."],"caption":"Conceptual procedure. Task variants and protocol versions retain their separate scoring conditions.","source_ids":["src-discovery-darlednik-geneb"],"source_locator":"Pinned README: Overview; data regimes; leaderboard aggregation definition"},"coverage":"limited","gaps":["Leakage controls: The protocol fixes preprocessing, partitions and seeds across models, but the inspected paper does not establish a benchmark-wide homology exclusion or pretraining-contamination audit. Shared evaluation settings are not proof that training corpora exclude test sequences."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary paper and/or task implementation reviewed for the explicitly cited methodology claims. Scope-limited absence is recorded only after the documented source search; no model runs or independent reproduction."}}},"description":"Frozen genomic representations with linear probes","facets":{"areas":["genomics"]},"id":"discovery-benchmark-geneb","kind":"benchmark","links":[],"name":"GENEB","source_ids":["src-discovery-darlednik-geneb"],"status":"discovered"} {"attributes":{"entity_level":"suite","scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Genomic sequence classification","version":null,"historical_missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"Genomic Benchmarks packages genomic sequence-classification datasets with explicit versions and supplied train/test folders.","summary_source_ids":["src-discovery-ml-bioinfo-ceitec-genomic-benchmarks"],"summary_source_locator":"Pinned README: repository purpose; info and download_dataset examples","sections":[{"title":"Evaluation methodology","body":"Genomic Benchmarks packages sequence-classification datasets together with versioned genomic coordinates, construction notebooks and a small CNN baseline. Each dataset supplies its own train and test subsets. Dataset-specific negative sampling and split provenance matter as much as model architecture for interpreting performance.","source_ids":["evidence-discovery-final-genomic-benchmarks"],"source_locator":"Methods: Reproducibility and Baseline model; Table 2"}],"facts":[{"label":"Datasets","value":"Named genomic classification datasets; the README illustrates a non-TATA human-promoter task.","status":"source_checked","source_ids":["src-discovery-ml-bioinfo-ceitec-genomic-benchmarks"],"source_locator":"Pinned README: repository purpose; info and download_dataset examples"},{"label":"Splits","value":"The download API delivers prepartitioned train/test data organized by class.","status":"source_checked","source_ids":["src-discovery-ml-bioinfo-ceitec-genomic-benchmarks"],"source_locator":"Pinned README: repository purpose; info and download_dataset examples"},{"label":"Metrics","value":"The original paper reports classification accuracy and F1 for its CNN baseline on each dataset.","status":"source_checked","source_ids":["evidence-discovery-final-genomic-benchmarks"],"source_locator":"Methods: Reproducibility and Baseline model; Table 2"},{"label":"Baselines","value":"The repository provides neural-network training helpers and links experiment reports.","status":"source_checked","source_ids":["src-discovery-ml-bioinfo-ceitec-genomic-benchmarks"],"source_locator":"Pinned README: repository purpose; info and download_dataset examples"},{"label":"Leakage controls","value":"Dataset-construction notebooks use fixed seeds. Generated negative regions match positive lengths and are rejected if they overlap positives; this does not establish a suite-wide chromosome-held-out or homology-filtered partition.","status":"source_checked","source_ids":["evidence-discovery-final-genomic-benchmarks"],"source_locator":"Methods: Reproducibility and Baseline model; Table 2"},{"label":"Uncertainty","value":"Table 2 gives point estimates for PyTorch and TensorFlow baseline implementations. The inspected Methods do not specify repeated-training or bootstrap uncertainty for those values.","status":"unreported","source_ids":["evidence-discovery-final-genomic-benchmarks"],"source_locator":"Methods: Reproducibility and Baseline model; Table 2"},{"label":"Entity type","value":"Repository of genomic sequence classification datasets.","status":"source_checked","source_ids":["src-discovery-ml-bioinfo-ceitec-genomic-benchmarks"],"source_locator":"Pinned README: repository purpose; info and download_dataset examples"},{"label":"Organisms","value":"Dataset-specific organisms; a human non-TATA promoter dataset is documented in the README.","status":"source_checked","source_ids":["src-discovery-ml-bioinfo-ceitec-genomic-benchmarks"],"source_locator":"Pinned README: repository purpose; info and download_dataset examples"},{"label":"Assays","value":"Curated genomic classification labels from linked source datasets.","status":"source_checked","source_ids":["src-discovery-ml-bioinfo-ceitec-genomic-benchmarks"],"source_locator":"Pinned README: repository purpose; info and download_dataset examples"},{"label":"Allowed inputs","value":"Sequences and categorical labels through the dataset loader.","status":"source_checked","source_ids":["src-discovery-ml-bioinfo-ceitec-genomic-benchmarks"],"source_locator":"Pinned README: repository purpose; info and download_dataset examples"},{"label":"Adaptation","value":"Supervised train/test classification with a documented CNN example.","status":"source_checked","source_ids":["src-discovery-ml-bioinfo-ceitec-genomic-benchmarks"],"source_locator":"Pinned README: repository purpose; info and download_dataset examples"}],"strengths":[{"text":"Common download/load conventions expose which named dataset a classifier used.","source_ids":["src-discovery-ml-bioinfo-ceitec-genomic-benchmarks"],"source_locator":"Pinned README: repository purpose; info and download_dataset examples"}],"limitations":[{"text":"Non-overlapping positive and negative genomic intervals do not guarantee independence between related sequences across the train/test boundary. Baseline implementation differences and data versions should remain explicit.","source_ids":["evidence-discovery-final-genomic-benchmarks"],"source_locator":"Methods: Reproducibility and Baseline model; Table 2"}],"diagram":{"title":"Evaluation procedure","steps":["Allowed inputs: Sequences and categorical labels through the dataset loader.","Splits: The download API delivers prepartitioned train/test data organized by class.","Metrics: The original paper reports classification accuracy and F1 for its CNN baseline on each dataset."],"caption":"Conceptual procedure. Task variants and protocol versions retain their separate scoring conditions.","source_ids":["src-discovery-ml-bioinfo-ceitec-genomic-benchmarks","evidence-discovery-final-genomic-benchmarks"],"source_locator":"Pinned README: repository purpose; info and download_dataset examples; Methods: Reproducibility and Baseline model; Table 2"},"coverage":"limited","gaps":["Uncertainty: Table 2 gives point estimates for PyTorch and TensorFlow baseline implementations. The inspected Methods do not specify repeated-training or bootstrap uncertainty for those values."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary paper and/or task implementation reviewed for the explicitly cited methodology claims. Scope-limited absence is recorded only after the documented source search; no model runs or independent reproduction."}}},"description":"Genomic sequence classification","facets":{"areas":["genomics"]},"id":"discovery-benchmark-genomic-benchmarks","kind":"benchmark","links":[],"name":"Genomic Benchmarks","source_ids":["src-discovery-ml-bioinfo-ceitec-genomic-benchmarks"],"status":"discovered"} {"attributes":{"entity_level":"suite","scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Glycan properties, taxonomy and molecular interactions","version":null,"historical_missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"GlycanML evaluates glycan learning across multiple classification and interaction tasks.","summary_source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-species-bert-yaml","evidence-benchmark-glycanml-bert-immunogenicity-bert-yaml","evidence-benchmark-glycanml-bert-link-bert-yaml","evidence-benchmark-glycanml-bert-interaction-bert-yaml"],"summary_source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned single-task BERT configurations for species, link, immunogenicity and interaction","sections":[{"title":"Evaluation methodology","body":"GlycanML evaluates taxonomy, immunogenicity, glycosylation type and protein–glycan interaction. Glycans are encoded as IUPAC sequences or graphs. Structural motif clusters define the first three task partitions, whereas interaction prediction holds out protein sequence clusters and predicts a transformed fluorescence binding signal.","source_ids":["evidence-discovery-final-glycanml"],"source_locator":"Sections 3.1–3.4"}],"facts":[{"label":"Datasets","value":"Glycan taxonomy, immunogenicity, glycosylation-type and protein–glycan interaction datasets.","status":"source_checked","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-species-bert-yaml","evidence-benchmark-glycanml-bert-immunogenicity-bert-yaml","evidence-benchmark-glycanml-bert-link-bert-yaml","evidence-benchmark-glycanml-bert-interaction-bert-yaml"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned single-task BERT configurations for species, link, immunogenicity and interaction"},{"label":"Splits","value":"Taxonomy, immunogenicity and glycosylation tasks use motif-cluster partitions. Protein–glycan interaction uses MMseqs2 protein clusters with threshold 0.5. Both allocate clusters 8:1:1; the two notions of held-out data are different.","status":"source_checked","source_ids":["evidence-discovery-final-glycanml"],"source_locator":"Sections 3.1–3.4"},{"label":"Metrics","value":"Checked single-task configurations use accuracy/MCC for taxonomy and glycosylation type, AUROC/AUPRC for immunogenicity, and MAE/RMSE/Spearman for protein–glycan interaction regression.","status":"source_checked","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-species-bert-yaml","evidence-benchmark-glycanml-bert-immunogenicity-bert-yaml","evidence-benchmark-glycanml-bert-link-bert-yaml","evidence-benchmark-glycanml-bert-interaction-bert-yaml"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned single-task BERT configurations for species, link, immunogenicity and interaction"},{"label":"Baselines","value":"Sequence-model CNN, ResNet, LSTM and BERT configurations; graph-model GCN, RGCN, GAT, GIN, CompGCN and MPNN configurations.","status":"source_checked","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-species-bert-yaml","evidence-benchmark-glycanml-bert-immunogenicity-bert-yaml","evidence-benchmark-glycanml-bert-link-bert-yaml","evidence-benchmark-glycanml-bert-interaction-bert-yaml"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned single-task BERT configurations for species, link, immunogenicity and interaction"},{"label":"Leakage controls","value":"Motif-based cluster separation tests transfer to structurally different glycans. This is a glycan-structure control, not a claim that all organisms or source studies are held out.","status":"source_checked","source_ids":["evidence-discovery-final-glycanml"],"source_locator":"Sections 3.1–3.3; Table 1"},{"label":"Uncertainty","value":"Every experiment uses seeds 0, 1 and 2; reported summaries are the mean and standard deviation over those three runs.","status":"source_checked","source_ids":["evidence-discovery-final-glycanml"],"source_locator":"Sections 3.1–3.4 and 5.1; Tables 1 and 3"},{"label":"Entity type","value":"Glycan representation benchmark suite.","status":"source_checked","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-species-bert-yaml","evidence-benchmark-glycanml-bert-immunogenicity-bert-yaml","evidence-benchmark-glycanml-bert-link-bert-yaml","evidence-benchmark-glycanml-bert-interaction-bert-yaml"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned single-task BERT configurations for species, link, immunogenicity and interaction"},{"label":"Organisms","value":"Taxonomy tasks explicitly predict organism categories; species scope depends on the constituent dataset.","status":"source_checked","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-species-bert-yaml","evidence-benchmark-glycanml-bert-immunogenicity-bert-yaml","evidence-benchmark-glycanml-bert-link-bert-yaml","evidence-benchmark-glycanml-bert-interaction-bert-yaml"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned single-task BERT configurations for species, link, immunogenicity and interaction"},{"label":"Assays","value":"Taxonomy, immunogenicity, glycosylation-type and protein–glycan interaction annotations.","status":"source_checked","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-species-bert-yaml","evidence-benchmark-glycanml-bert-immunogenicity-bert-yaml","evidence-benchmark-glycanml-bert-link-bert-yaml","evidence-benchmark-glycanml-bert-interaction-bert-yaml"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned single-task BERT configurations for species, link, immunogenicity and interaction"},{"label":"Allowed inputs","value":"Glycan sequence or graph representations and, for interaction tasks, paired protein data.","status":"source_checked","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-species-bert-yaml","evidence-benchmark-glycanml-bert-immunogenicity-bert-yaml","evidence-benchmark-glycanml-bert-link-bert-yaml","evidence-benchmark-glycanml-bert-interaction-bert-yaml"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned single-task BERT configurations for species, link, immunogenicity and interaction"},{"label":"Adaptation","value":"Separate single-task and multi-task training configurations are supplied.","status":"source_checked","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-species-bert-yaml","evidence-benchmark-glycanml-bert-immunogenicity-bert-yaml","evidence-benchmark-glycanml-bert-link-bert-yaml","evidence-benchmark-glycanml-bert-interaction-bert-yaml"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned single-task BERT configurations for species, link, immunogenicity and interaction"}],"strengths":[{"text":"Shared configurations allow controlled comparisons between single-task and multi-task learning.","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-species-bert-yaml","evidence-benchmark-glycanml-bert-immunogenicity-bert-yaml","evidence-benchmark-glycanml-bert-link-bert-yaml","evidence-benchmark-glycanml-bert-interaction-bert-yaml"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned single-task BERT configurations for species, link, immunogenicity and interaction"}],"limitations":[{"text":"Taxonomy, immunogenicity and glycosylation tasks hold out glycan motif clusters; interaction prediction holds out protein clusters. These boundaries do not imply that both proteins and glycans are unseen in the interaction task.","source_ids":["evidence-discovery-final-glycanml"],"source_locator":"Sections 3.1–3.4 and 5.1; Tables 1 and 3"}],"diagram":{"title":"Evaluation procedure","steps":["Allowed inputs: Glycan sequence or graph representations and, for interaction tasks, paired protein data.","Splits: Taxonomy, immunogenicity and glycosylation tasks use motif-cluster partitions. Protein–glycan interaction uses MMseqs2 protein clusters with threshold 0.5. Both allocate clusters 8:1:1; the two notions of held-out data are different.","Metrics: Checked single-task configurations use accuracy/MCC for taxonomy and glycosylation type, AUROC/AUPRC for immunogenicity, and MAE/RMSE/Spearman for protein–glycan interaction regression."],"caption":"Conceptual procedure. Task variants and protocol versions retain their separate scoring conditions.","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-species-bert-yaml","evidence-benchmark-glycanml-bert-immunogenicity-bert-yaml","evidence-benchmark-glycanml-bert-link-bert-yaml","evidence-benchmark-glycanml-bert-interaction-bert-yaml","evidence-discovery-final-glycanml"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned single-task BERT configurations for species, link, immunogenicity and interaction; Sections 3.1–3.4"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary paper and/or task implementation reviewed for the explicitly cited methodology claims. Scope-limited absence is recorded only after the documented source search; no model runs or independent reproduction."}}},"description":"Glycan properties, taxonomy and molecular interactions","facets":{"areas":["glycomics"]},"id":"discovery-benchmark-glycanml","kind":"benchmark","links":[],"name":"GlycanML","source_ids":["src-discovery-glycanml-glycanml"],"status":"discovered"} {"attributes":{"entity_level":"task","scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"glycosylation type prediction","version":null,"historical_missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"This GlycanML task evaluates glycosylation-type classification using glycan representations.","summary_source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-link-bert-yaml","evidence-benchmark-glycan-link-dataset"],"summary_source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned configs/single_task/BERT/link_BERT.yaml, module/custom_datasets/glycan_link.py (task metric, dataset class and split methods)","sections":[{"title":"Evaluation methodology","body":"N-linked, O-linked and free glycan classes. Dataset loader retains the train/validation/test assignments supplied in the downloaded CSV; this inspection does not establish how those original assignments were constructed. Accuracy and Matthews correlation coefficient in the checked three-class configuration. Sequence-model CNN, ResNet, LSTM and BERT configurations; graph-model GCN, RGCN, GAT, GIN, CompGCN and MPNN configurations.","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-link-bert-yaml","evidence-benchmark-glycan-link-dataset"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned configs/single_task/BERT/link_BERT.yaml, module/custom_datasets/glycan_link.py (task metric, dataset class and split methods)"}],"facts":[{"label":"Datasets","value":"N-linked, O-linked and free glycan classes.","status":"source_checked","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-link-bert-yaml","evidence-benchmark-glycan-link-dataset"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned configs/single_task/BERT/link_BERT.yaml, module/custom_datasets/glycan_link.py (task metric, dataset class and split methods)"},{"label":"Splits","value":"Glycans are represented by motif frequencies and clustered; motif groups are allocated to training, validation and test in an 8:1:1 grouping scheme. Table 1 preserves the resulting dataset-specific counts.","status":"source_checked","source_ids":["evidence-discovery-final-glycanml"],"source_locator":"Sections 3.1–3.3; Table 1"},{"label":"Metrics","value":"Accuracy and Matthews correlation coefficient in the checked three-class configuration.","status":"source_checked","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-link-bert-yaml","evidence-benchmark-glycan-link-dataset"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned configs/single_task/BERT/link_BERT.yaml, module/custom_datasets/glycan_link.py (task metric, dataset class and split methods)"},{"label":"Baselines","value":"Sequence-model CNN, ResNet, LSTM and BERT configurations; graph-model GCN, RGCN, GAT, GIN, CompGCN and MPNN configurations.","status":"source_checked","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-link-bert-yaml","evidence-benchmark-glycan-link-dataset"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned configs/single_task/BERT/link_BERT.yaml, module/custom_datasets/glycan_link.py (task metric, dataset class and split methods)"},{"label":"Leakage controls","value":"Motif-based cluster separation tests transfer to structurally different glycans. This is a glycan-structure control, not a claim that all organisms or source studies are held out.","status":"source_checked","source_ids":["evidence-discovery-final-glycanml"],"source_locator":"Sections 3.1–3.3; Table 1"},{"label":"Uncertainty","value":"Every experiment uses seeds 0, 1 and 2; reported summaries are the mean and standard deviation over those three runs.","status":"source_checked","source_ids":["evidence-discovery-final-glycanml"],"source_locator":"Sections 3.1–3.4 and 5.1; Tables 1 and 3"},{"label":"Entity type","value":"Constituent benchmark task: GlycanML glycosylation type prediction","status":"source_checked","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-link-bert-yaml","evidence-benchmark-glycan-link-dataset"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned configs/single_task/BERT/link_BERT.yaml, module/custom_datasets/glycan_link.py (task metric, dataset class and split methods)"},{"label":"Organisms","value":"Taxonomy tasks explicitly predict organism categories; species scope depends on the constituent dataset.","status":"source_checked","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-link-bert-yaml","evidence-benchmark-glycan-link-dataset"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned configs/single_task/BERT/link_BERT.yaml, module/custom_datasets/glycan_link.py (task metric, dataset class and split methods)"},{"label":"Assays","value":"Taxonomy, immunogenicity, glycosylation-type and protein–glycan interaction annotations.","status":"source_checked","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-link-bert-yaml","evidence-benchmark-glycan-link-dataset"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned configs/single_task/BERT/link_BERT.yaml, module/custom_datasets/glycan_link.py (task metric, dataset class and split methods)"},{"label":"Allowed inputs","value":"Glycan sequence or graph representation with a glycosylation-type target.","status":"source_checked","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-link-bert-yaml","evidence-benchmark-glycan-link-dataset"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned configs/single_task/BERT/link_BERT.yaml, module/custom_datasets/glycan_link.py (task metric, dataset class and split methods)"},{"label":"Adaptation","value":"Separate single-task and multi-task training configurations are supplied.","status":"source_checked","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-link-bert-yaml","evidence-benchmark-glycan-link-dataset"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned configs/single_task/BERT/link_BERT.yaml, module/custom_datasets/glycan_link.py (task metric, dataset class and split methods)"}],"strengths":[{"text":"Shared configurations allow controlled comparisons between single-task and multi-task learning.","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-link-bert-yaml","evidence-benchmark-glycan-link-dataset"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned configs/single_task/BERT/link_BERT.yaml, module/custom_datasets/glycan_link.py (task metric, dataset class and split methods)"}],"limitations":[{"text":"Taxonomy, immunogenicity and glycosylation tasks hold out glycan motif clusters; interaction prediction holds out protein clusters. These boundaries do not imply that both proteins and glycans are unseen in the interaction task.","source_ids":["evidence-discovery-final-glycanml"],"source_locator":"Sections 3.1–3.4 and 5.1; Tables 1 and 3"}],"diagram":{"title":"Evaluation procedure","steps":["Allowed inputs: Glycan sequence or graph representation with a glycosylation-type target.","Splits: Glycans are represented by motif frequencies and clustered; motif groups are allocated to training, validation and test in an 8:1:1 grouping scheme. Table 1 preserves the resulting dataset-specific counts.","Metrics: Accuracy and Matthews correlation coefficient in the checked three-class configuration."],"caption":"Conceptual procedure. Task variants and protocol versions retain their separate scoring conditions.","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-link-bert-yaml","evidence-benchmark-glycan-link-dataset","evidence-discovery-final-glycanml"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned configs/single_task/BERT/link_BERT.yaml, module/custom_datasets/glycan_link.py (task metric, dataset class and split methods); Sections 3.1–3.3; Table 1"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary paper and/or task implementation reviewed for the explicitly cited methodology claims. Scope-limited absence is recorded only after the documented source search; no model runs or independent reproduction."}}},"description":"glycosylation type prediction","facets":{"areas":["glycomics"]},"id":"discovery-benchmark-glycanml-glycosylation-type-prediction","kind":"benchmark","links":[{"relation":"parent","target_id":"discovery-benchmark-glycanml"},{"relation":"part_of","target_id":"discovery-benchmark-glycanml"}],"name":"GlycanML glycosylation type prediction","source_ids":["src-discovery-glycanml-glycanml"],"status":"discovered"} {"attributes":{"entity_level":"task","scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"immunogenicity prediction","version":null,"historical_missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"This GlycanML task evaluates immunogenicity classification using glycan representations.","summary_source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-immunogenicity-bert-yaml","evidence-benchmark-glycan-immunogenicity-dataset"],"summary_source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned configs/single_task/BERT/immunogenicity_BERT.yaml, module/custom_datasets/glycan_immunogenicity.py (task metric, dataset class and split methods)","sections":[{"title":"Evaluation methodology","body":"Binary glycan immunogenicity annotations. Dataset loader retains the train/validation/test assignments supplied in the downloaded CSV; this inspection does not establish how those original assignments were constructed. AUROC and AUPRC in the checked binary-classification configuration. Sequence-model CNN, ResNet, LSTM and BERT configurations; graph-model GCN, RGCN, GAT, GIN, CompGCN and MPNN configurations.","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-immunogenicity-bert-yaml","evidence-benchmark-glycan-immunogenicity-dataset"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned configs/single_task/BERT/immunogenicity_BERT.yaml, module/custom_datasets/glycan_immunogenicity.py (task metric, dataset class and split methods)"}],"facts":[{"label":"Datasets","value":"Binary glycan immunogenicity annotations.","status":"source_checked","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-immunogenicity-bert-yaml","evidence-benchmark-glycan-immunogenicity-dataset"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned configs/single_task/BERT/immunogenicity_BERT.yaml, module/custom_datasets/glycan_immunogenicity.py (task metric, dataset class and split methods)"},{"label":"Splits","value":"Glycans are represented by motif frequencies and clustered; motif groups are allocated to training, validation and test in an 8:1:1 grouping scheme. Table 1 preserves the resulting dataset-specific counts.","status":"source_checked","source_ids":["evidence-discovery-final-glycanml"],"source_locator":"Sections 3.1–3.3; Table 1"},{"label":"Metrics","value":"AUROC and AUPRC in the checked binary-classification configuration.","status":"source_checked","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-immunogenicity-bert-yaml","evidence-benchmark-glycan-immunogenicity-dataset"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned configs/single_task/BERT/immunogenicity_BERT.yaml, module/custom_datasets/glycan_immunogenicity.py (task metric, dataset class and split methods)"},{"label":"Baselines","value":"Sequence-model CNN, ResNet, LSTM and BERT configurations; graph-model GCN, RGCN, GAT, GIN, CompGCN and MPNN configurations.","status":"source_checked","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-immunogenicity-bert-yaml","evidence-benchmark-glycan-immunogenicity-dataset"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned configs/single_task/BERT/immunogenicity_BERT.yaml, module/custom_datasets/glycan_immunogenicity.py (task metric, dataset class and split methods)"},{"label":"Leakage controls","value":"Motif-based cluster separation tests transfer to structurally different glycans. This is a glycan-structure control, not a claim that all organisms or source studies are held out.","status":"source_checked","source_ids":["evidence-discovery-final-glycanml"],"source_locator":"Sections 3.1–3.3; Table 1"},{"label":"Uncertainty","value":"Every experiment uses seeds 0, 1 and 2; reported summaries are the mean and standard deviation over those three runs.","status":"source_checked","source_ids":["evidence-discovery-final-glycanml"],"source_locator":"Sections 3.1–3.4 and 5.1; Tables 1 and 3"},{"label":"Entity type","value":"Constituent benchmark task: GlycanML immunogenicity prediction","status":"source_checked","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-immunogenicity-bert-yaml","evidence-benchmark-glycan-immunogenicity-dataset"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned configs/single_task/BERT/immunogenicity_BERT.yaml, module/custom_datasets/glycan_immunogenicity.py (task metric, dataset class and split methods)"},{"label":"Organisms","value":"Taxonomy tasks explicitly predict organism categories; species scope depends on the constituent dataset.","status":"source_checked","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-immunogenicity-bert-yaml","evidence-benchmark-glycan-immunogenicity-dataset"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned configs/single_task/BERT/immunogenicity_BERT.yaml, module/custom_datasets/glycan_immunogenicity.py (task metric, dataset class and split methods)"},{"label":"Assays","value":"Taxonomy, immunogenicity, glycosylation-type and protein–glycan interaction annotations.","status":"source_checked","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-immunogenicity-bert-yaml","evidence-benchmark-glycan-immunogenicity-dataset"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned configs/single_task/BERT/immunogenicity_BERT.yaml, module/custom_datasets/glycan_immunogenicity.py (task metric, dataset class and split methods)"},{"label":"Allowed inputs","value":"Glycan sequence or graph representation with an immunogenicity target.","status":"source_checked","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-immunogenicity-bert-yaml","evidence-benchmark-glycan-immunogenicity-dataset"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned configs/single_task/BERT/immunogenicity_BERT.yaml, module/custom_datasets/glycan_immunogenicity.py (task metric, dataset class and split methods)"},{"label":"Adaptation","value":"Separate single-task and multi-task training configurations are supplied.","status":"source_checked","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-immunogenicity-bert-yaml","evidence-benchmark-glycan-immunogenicity-dataset"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned configs/single_task/BERT/immunogenicity_BERT.yaml, module/custom_datasets/glycan_immunogenicity.py (task metric, dataset class and split methods)"}],"strengths":[{"text":"Shared configurations allow controlled comparisons between single-task and multi-task learning.","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-immunogenicity-bert-yaml","evidence-benchmark-glycan-immunogenicity-dataset"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned configs/single_task/BERT/immunogenicity_BERT.yaml, module/custom_datasets/glycan_immunogenicity.py (task metric, dataset class and split methods)"}],"limitations":[{"text":"Taxonomy, immunogenicity and glycosylation tasks hold out glycan motif clusters; interaction prediction holds out protein clusters. These boundaries do not imply that both proteins and glycans are unseen in the interaction task.","source_ids":["evidence-discovery-final-glycanml"],"source_locator":"Sections 3.1–3.4 and 5.1; Tables 1 and 3"}],"diagram":{"title":"Evaluation procedure","steps":["Allowed inputs: Glycan sequence or graph representation with an immunogenicity target.","Splits: Glycans are represented by motif frequencies and clustered; motif groups are allocated to training, validation and test in an 8:1:1 grouping scheme. Table 1 preserves the resulting dataset-specific counts.","Metrics: AUROC and AUPRC in the checked binary-classification configuration."],"caption":"Conceptual procedure. Task variants and protocol versions retain their separate scoring conditions.","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-immunogenicity-bert-yaml","evidence-benchmark-glycan-immunogenicity-dataset","evidence-discovery-final-glycanml"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned configs/single_task/BERT/immunogenicity_BERT.yaml, module/custom_datasets/glycan_immunogenicity.py (task metric, dataset class and split methods); Sections 3.1–3.3; Table 1"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary paper and/or task implementation reviewed for the explicitly cited methodology claims. Scope-limited absence is recorded only after the documented source search; no model runs or independent reproduction."}}},"description":"immunogenicity prediction","facets":{"areas":["glycomics"]},"id":"discovery-benchmark-glycanml-immunogenicity-prediction","kind":"benchmark","links":[{"relation":"parent","target_id":"discovery-benchmark-glycanml"},{"relation":"part_of","target_id":"discovery-benchmark-glycanml"}],"name":"GlycanML immunogenicity prediction","source_ids":["src-discovery-glycanml-glycanml"],"status":"discovered"} {"attributes":{"entity_level":"task","scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"protein-glycan interaction prediction","version":null,"historical_missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"This GlycanML task evaluates protein–glycan interaction prediction using glycan representations.","summary_source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-interaction-bert-yaml","evidence-benchmark-glycan-interaction-dataset"],"summary_source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned configs/single_task/BERT/interaction_BERT.yaml, module/custom_datasets/glycan_interaction.py (task metric, dataset class and split methods)","sections":[{"title":"Evaluation methodology","body":"Protein–glycan interaction targets; the checked task is regression, not binary classification. Dataset loader retains the train/validation/test assignments supplied in the downloaded CSV; this inspection does not establish how those original assignments were constructed. MAE, RMSE and Spearman correlation in the checked interaction-regression configuration. Sequence-model CNN, ResNet, LSTM and BERT configurations; graph-model GCN, RGCN, GAT, GIN, CompGCN and MPNN configurations.","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-interaction-bert-yaml","evidence-benchmark-glycan-interaction-dataset"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned configs/single_task/BERT/interaction_BERT.yaml, module/custom_datasets/glycan_interaction.py (task metric, dataset class and split methods)"}],"facts":[{"label":"Datasets","value":"Protein–glycan interaction targets; the checked task is regression, not binary classification.","status":"source_checked","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-interaction-bert-yaml","evidence-benchmark-glycan-interaction-dataset"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned configs/single_task/BERT/interaction_BERT.yaml, module/custom_datasets/glycan_interaction.py (task metric, dataset class and split methods)"},{"label":"Splits","value":"Dataset loader retains the train/validation/test assignments supplied in the downloaded CSV; this inspection does not establish how those original assignments were constructed.","status":"source_checked","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-interaction-bert-yaml","evidence-benchmark-glycan-interaction-dataset"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned configs/single_task/BERT/interaction_BERT.yaml, module/custom_datasets/glycan_interaction.py (task metric, dataset class and split methods)"},{"label":"Metrics","value":"MAE, RMSE and Spearman correlation in the checked interaction-regression configuration.","status":"source_checked","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-interaction-bert-yaml","evidence-benchmark-glycan-interaction-dataset"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned configs/single_task/BERT/interaction_BERT.yaml, module/custom_datasets/glycan_interaction.py (task metric, dataset class and split methods)"},{"label":"Baselines","value":"Sequence-model CNN, ResNet, LSTM and BERT configurations; graph-model GCN, RGCN, GAT, GIN, CompGCN and MPNN configurations.","status":"source_checked","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-interaction-bert-yaml","evidence-benchmark-glycan-interaction-dataset"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned configs/single_task/BERT/interaction_BERT.yaml, module/custom_datasets/glycan_interaction.py (task metric, dataset class and split methods)"},{"label":"Leakage controls","value":"MMseqs2 groups proteins at minimum within-cluster sequence identity 0.5; complete protein clusters are allocated 8:1:1, and protein–glycan pairs follow the protein split. The protocol targets unseen proteins, not necessarily unseen glycans.","status":"source_checked","source_ids":["evidence-discovery-final-glycanml"],"source_locator":"Section 3.4"},{"label":"Uncertainty","value":"Every experiment uses seeds 0, 1 and 2; reported summaries are the mean and standard deviation over those three runs.","status":"source_checked","source_ids":["evidence-discovery-final-glycanml"],"source_locator":"Sections 3.1–3.4 and 5.1; Tables 1 and 3"},{"label":"Entity type","value":"Constituent benchmark task: GlycanML protein-glycan interaction prediction","status":"source_checked","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-interaction-bert-yaml","evidence-benchmark-glycan-interaction-dataset"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned configs/single_task/BERT/interaction_BERT.yaml, module/custom_datasets/glycan_interaction.py (task metric, dataset class and split methods)"},{"label":"Organisms","value":"Taxonomy tasks explicitly predict organism categories; species scope depends on the constituent dataset.","status":"source_checked","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-interaction-bert-yaml","evidence-benchmark-glycan-interaction-dataset"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned configs/single_task/BERT/interaction_BERT.yaml, module/custom_datasets/glycan_interaction.py (task metric, dataset class and split methods)"},{"label":"Assays","value":"Taxonomy, immunogenicity, glycosylation-type and protein–glycan interaction annotations.","status":"source_checked","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-interaction-bert-yaml","evidence-benchmark-glycan-interaction-dataset"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned configs/single_task/BERT/interaction_BERT.yaml, module/custom_datasets/glycan_interaction.py (task metric, dataset class and split methods)"},{"label":"Allowed inputs","value":"Protein–glycan pairs with interaction targets.","status":"source_checked","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-interaction-bert-yaml","evidence-benchmark-glycan-interaction-dataset"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned configs/single_task/BERT/interaction_BERT.yaml, module/custom_datasets/glycan_interaction.py (task metric, dataset class and split methods)"},{"label":"Adaptation","value":"Separate single-task and multi-task training configurations are supplied.","status":"source_checked","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-interaction-bert-yaml","evidence-benchmark-glycan-interaction-dataset"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned configs/single_task/BERT/interaction_BERT.yaml, module/custom_datasets/glycan_interaction.py (task metric, dataset class and split methods)"}],"strengths":[{"text":"Shared configurations allow controlled comparisons between single-task and multi-task learning.","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-interaction-bert-yaml","evidence-benchmark-glycan-interaction-dataset"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned configs/single_task/BERT/interaction_BERT.yaml, module/custom_datasets/glycan_interaction.py (task metric, dataset class and split methods)"}],"limitations":[{"text":"Taxonomy, immunogenicity and glycosylation tasks hold out glycan motif clusters; interaction prediction holds out protein clusters. These boundaries do not imply that both proteins and glycans are unseen in the interaction task.","source_ids":["evidence-discovery-final-glycanml"],"source_locator":"Sections 3.1–3.4 and 5.1; Tables 1 and 3"}],"diagram":{"title":"Evaluation procedure","steps":["Allowed inputs: Protein–glycan pairs with interaction targets.","Splits: Dataset loader retains the train/validation/test assignments supplied in the downloaded CSV; this inspection does not establish how those original assignments were constructed.","Metrics: MAE, RMSE and Spearman correlation in the checked interaction-regression configuration."],"caption":"Conceptual procedure. Task variants and protocol versions retain their separate scoring conditions.","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-interaction-bert-yaml","evidence-benchmark-glycan-interaction-dataset"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned configs/single_task/BERT/interaction_BERT.yaml, module/custom_datasets/glycan_interaction.py (task metric, dataset class and split methods)"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary paper and/or task implementation reviewed for the explicitly cited methodology claims. Scope-limited absence is recorded only after the documented source search; no model runs or independent reproduction."}}},"description":"protein-glycan interaction prediction","facets":{"areas":["glycomics"]},"id":"discovery-benchmark-glycanml-protein-glycan-interaction-prediction","kind":"benchmark","links":[{"relation":"parent","target_id":"discovery-benchmark-glycanml"},{"relation":"part_of","target_id":"discovery-benchmark-glycanml"}],"name":"GlycanML protein-glycan interaction prediction","source_ids":["src-discovery-glycanml-glycanml"],"status":"discovered"} {"attributes":{"entity_level":"task","scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"taxonomy prediction","version":null,"historical_missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"This GlycanML task evaluates taxonomy classification using glycan representations.","summary_source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-species-bert-yaml","evidence-benchmark-glycan-classification-dataset"],"summary_source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned configs/single_task/BERT/species_BERT.yaml, module/custom_datasets/glycan_classification.py (task metric, dataset class and split methods)","sections":[{"title":"Evaluation methodology","body":"Hierarchical taxonomy labels span species, genus, family, order, class, phylum, kingdom and domain; the checked single-task configuration targets species. Dataset loader retains the train/validation/test assignments supplied in the downloaded CSV; this inspection does not establish how those original assignments were constructed. Accuracy and Matthews correlation coefficient in the checked species-classification configuration. Sequence-model CNN, ResNet, LSTM and BERT configurations; graph-model GCN, RGCN, GAT, GIN, CompGCN and MPNN configurations.","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-species-bert-yaml","evidence-benchmark-glycan-classification-dataset"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned configs/single_task/BERT/species_BERT.yaml, module/custom_datasets/glycan_classification.py (task metric, dataset class and split methods)"}],"facts":[{"label":"Datasets","value":"Hierarchical taxonomy labels span species, genus, family, order, class, phylum, kingdom and domain; the checked single-task configuration targets species.","status":"source_checked","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-species-bert-yaml","evidence-benchmark-glycan-classification-dataset"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned configs/single_task/BERT/species_BERT.yaml, module/custom_datasets/glycan_classification.py (task metric, dataset class and split methods)"},{"label":"Splits","value":"Glycans are represented by motif frequencies and clustered; motif groups are allocated to training, validation and test in an 8:1:1 grouping scheme. Table 1 preserves the resulting dataset-specific counts.","status":"source_checked","source_ids":["evidence-discovery-final-glycanml"],"source_locator":"Sections 3.1–3.3; Table 1"},{"label":"Metrics","value":"Accuracy and Matthews correlation coefficient in the checked species-classification configuration.","status":"source_checked","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-species-bert-yaml","evidence-benchmark-glycan-classification-dataset"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned configs/single_task/BERT/species_BERT.yaml, module/custom_datasets/glycan_classification.py (task metric, dataset class and split methods)"},{"label":"Baselines","value":"Sequence-model CNN, ResNet, LSTM and BERT configurations; graph-model GCN, RGCN, GAT, GIN, CompGCN and MPNN configurations.","status":"source_checked","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-species-bert-yaml","evidence-benchmark-glycan-classification-dataset"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned configs/single_task/BERT/species_BERT.yaml, module/custom_datasets/glycan_classification.py (task metric, dataset class and split methods)"},{"label":"Leakage controls","value":"Motif-based cluster separation tests transfer to structurally different glycans. This is a glycan-structure control, not a claim that all organisms or source studies are held out.","status":"source_checked","source_ids":["evidence-discovery-final-glycanml"],"source_locator":"Sections 3.1–3.3; Table 1"},{"label":"Uncertainty","value":"Every experiment uses seeds 0, 1 and 2; reported summaries are the mean and standard deviation over those three runs.","status":"source_checked","source_ids":["evidence-discovery-final-glycanml"],"source_locator":"Sections 3.1–3.4 and 5.1; Tables 1 and 3"},{"label":"Entity type","value":"Constituent benchmark task: GlycanML taxonomy prediction","status":"source_checked","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-species-bert-yaml","evidence-benchmark-glycan-classification-dataset"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned configs/single_task/BERT/species_BERT.yaml, module/custom_datasets/glycan_classification.py (task metric, dataset class and split methods)"},{"label":"Organisms","value":"Taxonomy tasks explicitly predict organism categories; species scope depends on the constituent dataset.","status":"source_checked","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-species-bert-yaml","evidence-benchmark-glycan-classification-dataset"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned configs/single_task/BERT/species_BERT.yaml, module/custom_datasets/glycan_classification.py (task metric, dataset class and split methods)"},{"label":"Assays","value":"Taxonomy, immunogenicity, glycosylation-type and protein–glycan interaction annotations.","status":"source_checked","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-species-bert-yaml","evidence-benchmark-glycan-classification-dataset"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned configs/single_task/BERT/species_BERT.yaml, module/custom_datasets/glycan_classification.py (task metric, dataset class and split methods)"},{"label":"Allowed inputs","value":"Glycan sequence or graph representation with an organism-taxonomy target.","status":"source_checked","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-species-bert-yaml","evidence-benchmark-glycan-classification-dataset"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned configs/single_task/BERT/species_BERT.yaml, module/custom_datasets/glycan_classification.py (task metric, dataset class and split methods)"},{"label":"Adaptation","value":"Separate single-task and multi-task training configurations are supplied.","status":"source_checked","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-species-bert-yaml","evidence-benchmark-glycan-classification-dataset"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned configs/single_task/BERT/species_BERT.yaml, module/custom_datasets/glycan_classification.py (task metric, dataset class and split methods)"}],"strengths":[{"text":"Shared configurations allow controlled comparisons between single-task and multi-task learning.","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-species-bert-yaml","evidence-benchmark-glycan-classification-dataset"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned configs/single_task/BERT/species_BERT.yaml, module/custom_datasets/glycan_classification.py (task metric, dataset class and split methods)"}],"limitations":[{"text":"Taxonomy, immunogenicity and glycosylation tasks hold out glycan motif clusters; interaction prediction holds out protein clusters. These boundaries do not imply that both proteins and glycans are unseen in the interaction task.","source_ids":["evidence-discovery-final-glycanml"],"source_locator":"Sections 3.1–3.4 and 5.1; Tables 1 and 3"}],"diagram":{"title":"Evaluation procedure","steps":["Allowed inputs: Glycan sequence or graph representation with an organism-taxonomy target.","Splits: Glycans are represented by motif frequencies and clustered; motif groups are allocated to training, validation and test in an 8:1:1 grouping scheme. Table 1 preserves the resulting dataset-specific counts.","Metrics: Accuracy and Matthews correlation coefficient in the checked species-classification configuration."],"caption":"Conceptual procedure. Task variants and protocol versions retain their separate scoring conditions.","source_ids":["src-discovery-glycanml-glycanml","evidence-benchmark-glycanml-bert-species-bert-yaml","evidence-benchmark-glycan-classification-dataset","evidence-discovery-final-glycanml"],"source_locator":"Pinned README: Introduction; Experiment configurations; leaderboard; pinned configs/single_task/BERT/species_BERT.yaml, module/custom_datasets/glycan_classification.py (task metric, dataset class and split methods); Sections 3.1–3.3; Table 1"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary paper and/or task implementation reviewed for the explicitly cited methodology claims. Scope-limited absence is recorded only after the documented source search; no model runs or independent reproduction."}}},"description":"taxonomy prediction","facets":{"areas":["glycomics"]},"id":"discovery-benchmark-glycanml-taxonomy-prediction","kind":"benchmark","links":[{"relation":"parent","target_id":"discovery-benchmark-glycanml"},{"relation":"part_of","target_id":"discovery-benchmark-glycanml"}],"name":"GlycanML taxonomy prediction","source_ids":["src-discovery-glycanml-glycanml"],"status":"discovered"} {"attributes":{"entity_level":"suite","scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Multi-species genome understanding tasks","version":null,"historical_missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"GUE evaluates genome understanding across multiple datasets, task types and species.","summary_source_ids":["src-discovery-magics-lab-dnabert-2"],"summary_source_locator":"Pinned README: GUE section; data download and evaluation scripts","sections":[{"title":"Evaluation methodology","body":"GUE combines genome sequence classification tasks with supplied partitions and task-specific metrics. Models are fine-tuned on labelled training examples, selected with validation data and scored on test data. Random partitions occur in the suite, so a high score does not automatically demonstrate transfer to unrelated genomes.","source_ids":["evidence-discovery-final-gue"],"source_locator":"Section 5; Appendix C and Table 9: GUE task datasets"}],"facts":[{"label":"Datasets","value":"The benchmark archive contains separate sequence-classification datasets; model pretraining data are distributed separately.","status":"source_checked","source_ids":["src-discovery-magics-lab-dnabert-2"],"source_locator":"Pinned README: GUE section; data download and evaluation scripts"},{"label":"Splits","value":"Separate train/validation/test files and counts are defined per dataset. The yeast epigenetic tasks use an 8:1:1 random split; source-derived regulatory and viral tasks retain their own documented partitions. There is no universal chromosome holdout across GUE.","status":"source_checked","source_ids":["evidence-discovery-final-gue"],"source_locator":"Appendix C: task construction; Table 9"},{"label":"Metrics","value":"GUE uses Matthews correlation coefficient for the regulatory classification tasks and F1 for COVID variant classification; the exact task metric is tabulated in the dataset appendix.","status":"source_checked","source_ids":["evidence-discovery-final-gue"],"source_locator":"Section 5; Appendix C and Table 9: GUE task datasets"},{"label":"Baselines","value":"DNABERT-2 and other genomic representation models are compared in the associated benchmark.","status":"source_checked","source_ids":["src-discovery-magics-lab-dnabert-2"],"source_locator":"Pinned README: GUE section; data download and evaluation scripts"},{"label":"Leakage controls","value":"Datasets have explicit train/validation/test partitions, but some use random splits, including yeast epigenetic marks at 8:1:1. The paper’s masked-token leakage discussion concerns tokenization and must not be confused with proof of train/test genome independence.","status":"source_checked","source_ids":["evidence-discovery-final-gue"],"source_locator":"Section 5; Appendix C and Table 9: GUE task datasets"},{"label":"Uncertainty","value":"Models are fine-tuned with three different random seeds and the mean test result is reported. The stated protocol does not define a uniform confidence interval for every dataset.","status":"source_checked","source_ids":["evidence-discovery-final-gue"],"source_locator":"Section 5; Appendix C and Table 9: GUE task datasets"},{"label":"Entity type","value":"Genome Understanding Evaluation benchmark suite.","status":"source_checked","source_ids":["src-discovery-magics-lab-dnabert-2"],"source_locator":"Pinned README: GUE section; data download and evaluation scripts"},{"label":"Organisms","value":"Multiple species; the README describes four-species coverage.","status":"source_checked","source_ids":["src-discovery-magics-lab-dnabert-2"],"source_locator":"Pinned README: GUE section; data download and evaluation scripts"},{"label":"Assays","value":"Task-specific genomic classification labels.","status":"source_checked","source_ids":["src-discovery-magics-lab-dnabert-2"],"source_locator":"Pinned README: GUE section; data download and evaluation scripts"},{"label":"Allowed inputs","value":"DNA sequences from the benchmark archive, separate from model-pretraining data.","status":"source_checked","source_ids":["src-discovery-magics-lab-dnabert-2"],"source_locator":"Pinned README: GUE section; data download and evaluation scripts"},{"label":"Adaptation","value":"Supervised fine-tuning; scripts include model-specific training examples.","status":"source_checked","source_ids":["src-discovery-magics-lab-dnabert-2"],"source_locator":"Pinned README: GUE section; data download and evaluation scripts"}],"strengths":[{"text":"The benchmark distribution is separated from the pretraining corpus.","source_ids":["src-discovery-magics-lab-dnabert-2"],"source_locator":"Pinned README: GUE section; data download and evaluation scripts"}],"limitations":[{"text":"Dataset partitions and metrics are task-specific. Non-overlapping tokenization addresses masked-token information leakage, which is a different issue from biological homology or pretraining overlap.","source_ids":["evidence-discovery-final-gue"],"source_locator":"Section 5; Appendix C and Table 9: GUE task datasets"}],"diagram":{"title":"Evaluation procedure","steps":["Allowed inputs: DNA sequences from the benchmark archive, separate from model-pretraining data.","Splits: Separate train/validation/test files and counts are defined per dataset. The yeast epigenetic tasks use an 8:1:1 random split; source-derived regulatory and viral tasks retain their own documented partitions. There is no universal chromosome holdout across GUE.","Metrics: GUE uses Matthews correlation coefficient for the regulatory classification tasks and F1 for COVID variant classification; the exact task metric is tabulated in the dataset appendix."],"caption":"Conceptual procedure. Task variants and protocol versions retain their separate scoring conditions.","source_ids":["src-discovery-magics-lab-dnabert-2","evidence-discovery-final-gue"],"source_locator":"Pinned README: GUE section; data download and evaluation scripts; Appendix C: task construction; Table 9; Section 5; Appendix C and Table 9: GUE task datasets"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary paper and/or task implementation reviewed for the explicitly cited methodology claims. Scope-limited absence is recorded only after the documented source search; no model runs or independent reproduction."}}},"description":"Multi-species genome understanding tasks","facets":{"areas":["genomics"]},"id":"discovery-benchmark-gue","kind":"benchmark","links":[],"name":"GUE","source_ids":["src-discovery-magics-lab-dnabert-2"],"status":"discovered"} {"attributes":{"entity_level":"suite","scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Gene expression prediction from matched histology","version":null,"historical_missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"HEST-Benchmark tests prediction of gene expression from histological image representations.","summary_source_ids":["src-discovery-mahmoodlab-hest"],"summary_source_locator":"Pinned README: HEST-Benchmark overview; evaluation notes","sections":[{"title":"Evaluation methodology","body":"HEST-Benchmark connects histology patches to spatially measured gene expression. A patch encoder supplies features to a regression model, which predicts highly variable genes. Patient-stratified folds evaluate transfer between individuals, and correlation is summarized across those folds.","source_ids":["evidence-discovery-final-hest"],"source_locator":"Sections 5.1–5.2; Appendix Table A11"}],"facts":[{"label":"Datasets","value":"Paired spatial-transcriptomic measurements and histology images from HEST resources.","status":"source_checked","source_ids":["src-discovery-mahmoodlab-hest"],"source_locator":"Pinned README: HEST-Benchmark overview; evaluation notes"},{"label":"Splits","value":"Patient-stratified cross-validation: one fold per patient, except ccRCC uses half as many folds because of its larger patient cohort.","status":"source_checked","source_ids":["evidence-discovery-final-hest"],"source_locator":"Sections 5.1–5.2; Appendix Table A11"},{"label":"Metrics","value":"Pearson correlation between predicted and measured log1p gene expression, using the 50 genes with highest normalized variance.","status":"source_checked","source_ids":["evidence-discovery-final-hest"],"source_locator":"Sections 5.1–5.2; Appendix Table A11"},{"label":"Baselines","value":"Ridge regression on PCA-reduced embeddings is the reported comparison setup.","status":"source_checked","source_ids":["src-discovery-mahmoodlab-hest"],"source_locator":"Pinned README: HEST-Benchmark overview; evaluation notes"},{"label":"Leakage controls","value":"All samples from a patient stay within their fold, preventing patch-level mixing of the same patient between training and test. Histology-encoder pretraining overlap is a separate concern.","status":"source_checked","source_ids":["evidence-discovery-final-hest"],"source_locator":"Sections 5.1–5.2; Appendix Table A11"},{"label":"Uncertainty","value":"The paper reports the mean and standard deviation across folds or patients, not a universal retraining-seed interval.","status":"source_checked","source_ids":["evidence-discovery-final-hest"],"source_locator":"Sections 5.1–5.2; Appendix Table A11"},{"label":"Entity type","value":"Spatial-transcriptomics prediction benchmark.","status":"source_checked","source_ids":["src-discovery-mahmoodlab-hest"],"source_locator":"Pinned README: HEST-Benchmark overview; evaluation notes"},{"label":"Organisms","value":"Multiple species selectable in the HEST metadata.","status":"source_checked","source_ids":["src-discovery-mahmoodlab-hest"],"source_locator":"Pinned README: HEST-Benchmark overview; evaluation notes"},{"label":"Assays","value":"Paired tissue histology and spatial gene-expression measurements.","status":"source_checked","source_ids":["src-discovery-mahmoodlab-hest"],"source_locator":"Pinned README: HEST-Benchmark overview; evaluation notes"},{"label":"Allowed inputs","value":"Histology patches for prediction; spatial expression supplies evaluation labels.","status":"source_checked","source_ids":["src-discovery-mahmoodlab-hest"],"source_locator":"Pinned README: HEST-Benchmark overview; evaluation notes"},{"label":"Adaptation","value":"Patch embeddings are evaluated through downstream expression prediction.","status":"source_checked","source_ids":["src-discovery-mahmoodlab-hest"],"source_locator":"Pinned README: HEST-Benchmark overview; evaluation notes"}],"strengths":[{"text":"Paired measurements connect image representations to molecular rather than image-only labels.","source_ids":["src-discovery-mahmoodlab-hest"],"source_locator":"Pinned README: HEST-Benchmark overview; evaluation notes"}],"limitations":[{"text":"The input is an image, but the evaluated output is spatial molecular expression. The selected genes, regression head and patient partitions are essential comparison conditions.","source_ids":["evidence-discovery-final-hest"],"source_locator":"Sections 5.1–5.2; Appendix Table A11"}],"diagram":{"title":"Evaluation procedure","steps":["Allowed inputs: Histology patches for prediction; spatial expression supplies evaluation labels.","Splits: Patient-stratified cross-validation: one fold per patient, except ccRCC uses half as many folds because of its larger patient cohort.","Metrics: Pearson correlation between predicted and measured log1p gene expression, using the 50 genes with highest normalized variance."],"caption":"Conceptual procedure. Task variants and protocol versions retain their separate scoring conditions.","source_ids":["src-discovery-mahmoodlab-hest","evidence-discovery-final-hest"],"source_locator":"Pinned README: HEST-Benchmark overview; evaluation notes; Sections 5.1–5.2; Appendix Table A11"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary paper and/or task implementation reviewed for the explicitly cited methodology claims. Scope-limited absence is recorded only after the documented source search; no model runs or independent reproduction."}}},"description":"Gene expression prediction from matched histology","facets":{"areas":["spatial-omics"]},"id":"discovery-benchmark-hest-benchmark","kind":"benchmark","links":[],"name":"HEST-Benchmark","source_ids":["src-discovery-mahmoodlab-hest"],"status":"discovered"} {"attributes":{"entity_level":"suite","scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Molecular identification from tandem mass spectra","version":null,"historical_missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"MassSpecGym separates spectrum-to-structure generation, candidate retrieval and structure-to-spectrum simulation.","summary_source_ids":["src-discovery-pluskal-lab-massspecgym","evidence-benchmark-massspecgym-de-novo-base-py","evidence-benchmark-massspecgym-retrieval-base-py","evidence-benchmark-massspecgym-simulation-base-py"],"summary_source_locator":"Pinned README: three challenges; dataset and DataModule; evaluation base classes; pinned de_novo, retrieval and simulation base evaluation classes","sections":[{"title":"Evaluation methodology","body":"The released MassSpecGym MS/MS/molecule dataset, with task-specific inputs and candidate sets. MCES molecular clusters are grouped into fixed training, validation and test folds, stratified by acquisition metadata. Cross-fold molecular bond-edit distance is at least 10; all spectra follow the assigned molecular fold. Task evaluators distinguish molecular exact match/structural similarity, candidate-retrieval hit rate and spectrum similarity. These are separate readouts, not interchangeable scores. Maximum common edge subgraph (MCES) clustering keeps molecules connected by a bond-edit distance below 10 in the same fold. The split additionally balances instrument, collision-energy, adduct and molecule-frequency metadata; this is stronger than simply separating 2D InChIKeys.","source_ids":["src-discovery-pluskal-lab-massspecgym","evidence-benchmark-massspecgym-de-novo-base-py","evidence-benchmark-massspecgym-retrieval-base-py","evidence-benchmark-massspecgym-simulation-base-py","evidence-discovery-final-massspecgym"],"source_locator":"Pinned README: three challenges; dataset and DataModule; evaluation base classes; pinned de_novo, retrieval and simulation base evaluation classes; Section 3.4; Supplementary Information 2.5; Section 3.4; Supplementary Information 2.5; Tables 2–4"}],"facts":[{"label":"Datasets","value":"The released MassSpecGym MS/MS/molecule dataset, with task-specific inputs and candidate sets.","status":"source_checked","source_ids":["src-discovery-pluskal-lab-massspecgym","evidence-benchmark-massspecgym-de-novo-base-py","evidence-benchmark-massspecgym-retrieval-base-py","evidence-benchmark-massspecgym-simulation-base-py"],"source_locator":"Pinned README: three challenges; dataset and DataModule; evaluation base classes; pinned de_novo, retrieval and simulation base evaluation classes"},{"label":"Splits","value":"MCES molecular clusters are grouped into fixed training, validation and test folds, stratified by acquisition metadata. Cross-fold molecular bond-edit distance is at least 10; all spectra follow the assigned molecular fold.","status":"source_checked","source_ids":["evidence-discovery-final-massspecgym"],"source_locator":"Section 3.4; Supplementary Information 2.5"},{"label":"Metrics","value":"Task evaluators distinguish molecular exact match/structural similarity, candidate-retrieval hit rate and spectrum similarity. These are separate readouts, not interchangeable scores.","status":"source_checked","source_ids":["src-discovery-pluskal-lab-massspecgym","evidence-benchmark-massspecgym-de-novo-base-py","evidence-benchmark-massspecgym-retrieval-base-py","evidence-benchmark-massspecgym-simulation-base-py"],"source_locator":"Pinned README: three challenges; dataset and DataModule; evaluation base classes; pinned de_novo, retrieval and simulation base evaluation classes"},{"label":"Baselines","value":"The README illustrates a DeepSets-style spectrum-to-fingerprint retrieval baseline.","status":"source_checked","source_ids":["src-discovery-pluskal-lab-massspecgym","evidence-benchmark-massspecgym-de-novo-base-py","evidence-benchmark-massspecgym-retrieval-base-py","evidence-benchmark-massspecgym-simulation-base-py"],"source_locator":"Pinned README: three challenges; dataset and DataModule; evaluation base classes; pinned de_novo, retrieval and simulation base evaluation classes"},{"label":"Leakage controls","value":"Maximum common edge subgraph (MCES) clustering keeps molecules connected by a bond-edit distance below 10 in the same fold. The split additionally balances instrument, collision-energy, adduct and molecule-frequency metadata; this is stronger than simply separating 2D InChIKeys.","status":"source_checked","source_ids":["evidence-discovery-final-massspecgym"],"source_locator":"Section 3.4; Supplementary Information 2.5; Tables 2–4"},{"label":"Uncertainty","value":"Tables 2–4 report 99.9% bootstrap confidence intervals using 20,000 resamples. These intervals summarize test-example sampling, not variation across independently retrained models.","status":"source_checked","source_ids":["evidence-discovery-final-massspecgym"],"source_locator":"Section 3.4; Supplementary Information 2.5; Tables 2–4"},{"label":"Entity type","value":"Small-molecule MS/MS benchmark with three task directions.","status":"source_checked","source_ids":["src-discovery-pluskal-lab-massspecgym","evidence-benchmark-massspecgym-de-novo-base-py","evidence-benchmark-massspecgym-retrieval-base-py","evidence-benchmark-massspecgym-simulation-base-py"],"source_locator":"Pinned README: three challenges; dataset and DataModule; evaluation base classes; pinned de_novo, retrieval and simulation base evaluation classes"},{"label":"Organisms","value":"Molecule identity rather than organism classification defines these tasks.","status":"inapplicable","source_ids":["src-discovery-pluskal-lab-massspecgym","evidence-benchmark-massspecgym-de-novo-base-py","evidence-benchmark-massspecgym-retrieval-base-py","evidence-benchmark-massspecgym-simulation-base-py"],"source_locator":"Pinned README: three challenges; dataset and DataModule; evaluation base classes; pinned de_novo, retrieval and simulation base evaluation classes"},{"label":"Assays","value":"Tandem mass spectra paired with molecular structures.","status":"source_checked","source_ids":["src-discovery-pluskal-lab-massspecgym","evidence-benchmark-massspecgym-de-novo-base-py","evidence-benchmark-massspecgym-retrieval-base-py","evidence-benchmark-massspecgym-simulation-base-py"],"source_locator":"Pinned README: three challenges; dataset and DataModule; evaluation base classes; pinned de_novo, retrieval and simulation base evaluation classes"},{"label":"Allowed inputs","value":"Spectrum-to-molecule, spectrum-plus-candidates, or molecule-to-spectrum inputs depend on the selected task.","status":"source_checked","source_ids":["src-discovery-pluskal-lab-massspecgym","evidence-benchmark-massspecgym-de-novo-base-py","evidence-benchmark-massspecgym-retrieval-base-py","evidence-benchmark-massspecgym-simulation-base-py"],"source_locator":"Pinned README: three challenges; dataset and DataModule; evaluation base classes; pinned de_novo, retrieval and simulation base evaluation classes"},{"label":"Adaptation","value":"Supervised train/validation/test learning; pretrained or new models use the task-specific interfaces.","status":"source_checked","source_ids":["src-discovery-pluskal-lab-massspecgym","evidence-benchmark-massspecgym-de-novo-base-py","evidence-benchmark-massspecgym-retrieval-base-py","evidence-benchmark-massspecgym-simulation-base-py"],"source_locator":"Pinned README: three challenges; dataset and DataModule; evaluation base classes; pinned de_novo, retrieval and simulation base evaluation classes"}],"strengths":[{"text":"One release supplies explicit task interfaces and predefined splits for three complementary predictions.","source_ids":["src-discovery-pluskal-lab-massspecgym","evidence-benchmark-massspecgym-de-novo-base-py","evidence-benchmark-massspecgym-retrieval-base-py","evidence-benchmark-massspecgym-simulation-base-py"],"source_locator":"Pinned README: three challenges; dataset and DataModule; evaluation base classes; pinned de_novo, retrieval and simulation base evaluation classes"}],"limitations":[{"text":"Chemical-formula-assisted tasks provide extra input information and must remain separate from unassisted tasks. The MCES split constrains structural similarity but cannot establish independence from every external pretraining corpus.","source_ids":["evidence-discovery-final-massspecgym"],"source_locator":"Section 3.4; Supplementary Information 2.5; Tables 2–4"}],"diagram":{"title":"Evaluation procedure","steps":["Allowed inputs: Spectrum-to-molecule, spectrum-plus-candidates, or molecule-to-spectrum inputs depend on the selected task.","Splits: MCES molecular clusters are grouped into fixed training, validation and test folds, stratified by acquisition metadata. Cross-fold molecular bond-edit distance is at least 10; all spectra follow the assigned molecular fold.","Metrics: Task evaluators distinguish molecular exact match/structural similarity, candidate-retrieval hit rate and spectrum similarity. These are separate readouts, not interchangeable scores."],"caption":"Conceptual procedure. Task variants and protocol versions retain their separate scoring conditions.","source_ids":["src-discovery-pluskal-lab-massspecgym","evidence-benchmark-massspecgym-de-novo-base-py","evidence-benchmark-massspecgym-retrieval-base-py","evidence-benchmark-massspecgym-simulation-base-py","evidence-discovery-final-massspecgym"],"source_locator":"Pinned README: three challenges; dataset and DataModule; evaluation base classes; pinned de_novo, retrieval and simulation base evaluation classes; Section 3.4; Supplementary Information 2.5"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary paper and/or task implementation reviewed for the explicitly cited methodology claims. Scope-limited absence is recorded only after the documented source search; no model runs or independent reproduction."}}},"description":"Molecular identification from tandem mass spectra","facets":{"areas":["metabolomics"]},"id":"discovery-benchmark-massspecgym","kind":"benchmark","links":[],"name":"MassSpecGym","source_ids":["src-discovery-pluskal-lab-massspecgym"],"status":"discovered"} {"attributes":{"entity_level":"task","scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Generate molecular structures from tandem mass spectra","version":null,"historical_missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"This task predicts molecular structures from an MS/MS spectrum.","summary_source_ids":["src-discovery-pluskal-lab-massspecgym","evidence-benchmark-massspecgym-de-novo-base-py"],"summary_source_locator":"Pinned README: three challenges; dataset and DataModule; evaluation base classes; pinned massspecgym/models/de_novo/base.py evaluation methods","sections":[{"title":"Evaluation methodology","body":"MS/MS spectrum input and molecular-structure target; formula-assisted variant is separate. MCES molecular clusters are grouped into fixed training, validation and test folds, stratified by acquisition metadata. Cross-fold molecular bond-edit distance is at least 10; all spectra follow the assigned molecular fold. Top-k molecular exact-match accuracy using InChIKey identity, maximum fingerprint Tanimoto similarity, minimum MCES distance and predicted-molecule validity. Maximum common edge subgraph (MCES) clustering keeps molecules connected by a bond-edit distance below 10 in the same fold. The split additionally balances instrument, collision-energy, adduct and molecule-frequency metadata; this is stronger than simply separating 2D InChIKeys.","source_ids":["src-discovery-pluskal-lab-massspecgym","evidence-benchmark-massspecgym-de-novo-base-py","evidence-discovery-final-massspecgym"],"source_locator":"Pinned README: three challenges; dataset and DataModule; evaluation base classes; pinned massspecgym/models/de_novo/base.py evaluation methods; Section 3.4; Supplementary Information 2.5; Section 3.4; Supplementary Information 2.5; Tables 2–4"}],"facts":[{"label":"Datasets","value":"MS/MS spectrum input and molecular-structure target; formula-assisted variant is separate.","status":"source_checked","source_ids":["src-discovery-pluskal-lab-massspecgym","evidence-benchmark-massspecgym-de-novo-base-py"],"source_locator":"Pinned README: three challenges; dataset and DataModule; evaluation base classes; pinned massspecgym/models/de_novo/base.py evaluation methods"},{"label":"Splits","value":"MCES molecular clusters are grouped into fixed training, validation and test folds, stratified by acquisition metadata. Cross-fold molecular bond-edit distance is at least 10; all spectra follow the assigned molecular fold.","status":"source_checked","source_ids":["evidence-discovery-final-massspecgym"],"source_locator":"Section 3.4; Supplementary Information 2.5"},{"label":"Metrics","value":"Top-k molecular exact-match accuracy using InChIKey identity, maximum fingerprint Tanimoto similarity, minimum MCES distance and predicted-molecule validity.","status":"source_checked","source_ids":["src-discovery-pluskal-lab-massspecgym","evidence-benchmark-massspecgym-de-novo-base-py"],"source_locator":"Pinned README: three challenges; dataset and DataModule; evaluation base classes; pinned massspecgym/models/de_novo/base.py evaluation methods"},{"label":"Baselines","value":"The README illustrates a DeepSets-style spectrum-to-fingerprint retrieval baseline.","status":"source_checked","source_ids":["src-discovery-pluskal-lab-massspecgym","evidence-benchmark-massspecgym-de-novo-base-py"],"source_locator":"Pinned README: three challenges; dataset and DataModule; evaluation base classes; pinned massspecgym/models/de_novo/base.py evaluation methods"},{"label":"Leakage controls","value":"Maximum common edge subgraph (MCES) clustering keeps molecules connected by a bond-edit distance below 10 in the same fold. The split additionally balances instrument, collision-energy, adduct and molecule-frequency metadata; this is stronger than simply separating 2D InChIKeys.","status":"source_checked","source_ids":["evidence-discovery-final-massspecgym"],"source_locator":"Section 3.4; Supplementary Information 2.5; Tables 2–4"},{"label":"Uncertainty","value":"Tables 2–4 report 99.9% bootstrap confidence intervals using 20,000 resamples. These intervals summarize test-example sampling, not variation across independently retrained models.","status":"source_checked","source_ids":["evidence-discovery-final-massspecgym"],"source_locator":"Section 3.4; Supplementary Information 2.5; Tables 2–4"},{"label":"Entity type","value":"Constituent benchmark task: MassSpecGym De novo molecule generation","status":"source_checked","source_ids":["src-discovery-pluskal-lab-massspecgym","evidence-benchmark-massspecgym-de-novo-base-py"],"source_locator":"Pinned README: three challenges; dataset and DataModule; evaluation base classes; pinned massspecgym/models/de_novo/base.py evaluation methods"},{"label":"Organisms","value":"Molecule identity rather than organism classification defines these tasks.","status":"inapplicable","source_ids":["src-discovery-pluskal-lab-massspecgym","evidence-benchmark-massspecgym-de-novo-base-py"],"source_locator":"Pinned README: three challenges; dataset and DataModule; evaluation base classes; pinned massspecgym/models/de_novo/base.py evaluation methods"},{"label":"Assays","value":"Tandem mass spectra paired with molecular structures.","status":"source_checked","source_ids":["src-discovery-pluskal-lab-massspecgym","evidence-benchmark-massspecgym-de-novo-base-py"],"source_locator":"Pinned README: three challenges; dataset and DataModule; evaluation base classes; pinned massspecgym/models/de_novo/base.py evaluation methods"},{"label":"Allowed inputs","value":"MS/MS spectrum; molecular formula is available only in the separately identified formula-assisted variant.","status":"source_checked","source_ids":["src-discovery-pluskal-lab-massspecgym","evidence-benchmark-massspecgym-de-novo-base-py"],"source_locator":"Pinned README: three challenges; dataset and DataModule; evaluation base classes; pinned massspecgym/models/de_novo/base.py evaluation methods"},{"label":"Adaptation","value":"Supervised train/validation/test learning; pretrained or new models use the task-specific interfaces.","status":"source_checked","source_ids":["src-discovery-pluskal-lab-massspecgym","evidence-benchmark-massspecgym-de-novo-base-py"],"source_locator":"Pinned README: three challenges; dataset and DataModule; evaluation base classes; pinned massspecgym/models/de_novo/base.py evaluation methods"}],"strengths":[{"text":"One release supplies explicit task interfaces and predefined splits for three complementary predictions.","source_ids":["src-discovery-pluskal-lab-massspecgym","evidence-benchmark-massspecgym-de-novo-base-py"],"source_locator":"Pinned README: three challenges; dataset and DataModule; evaluation base classes; pinned massspecgym/models/de_novo/base.py evaluation methods"}],"limitations":[{"text":"Chemical-formula-assisted tasks provide extra input information and must remain separate from unassisted tasks. The MCES split constrains structural similarity but cannot establish independence from every external pretraining corpus.","source_ids":["evidence-discovery-final-massspecgym"],"source_locator":"Section 3.4; Supplementary Information 2.5; Tables 2–4"}],"diagram":{"title":"Evaluation procedure","steps":["Allowed inputs: MS/MS spectrum; molecular formula is available only in the separately identified formula-assisted variant.","Splits: MCES molecular clusters are grouped into fixed training, validation and test folds, stratified by acquisition metadata. Cross-fold molecular bond-edit distance is at least 10; all spectra follow the assigned molecular fold.","Metrics: Top-k molecular exact-match accuracy using InChIKey identity, maximum fingerprint Tanimoto similarity, minimum MCES distance and predicted-molecule validity."],"caption":"Conceptual procedure. Task variants and protocol versions retain their separate scoring conditions.","source_ids":["src-discovery-pluskal-lab-massspecgym","evidence-benchmark-massspecgym-de-novo-base-py","evidence-discovery-final-massspecgym"],"source_locator":"Pinned README: three challenges; dataset and DataModule; evaluation base classes; pinned massspecgym/models/de_novo/base.py evaluation methods; Section 3.4; Supplementary Information 2.5"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary paper and/or task implementation reviewed for the explicitly cited methodology claims. Scope-limited absence is recorded only after the documented source search; no model runs or independent reproduction."}}},"description":"Generate molecular structures from tandem mass spectra","facets":{"areas":["metabolomics"]},"id":"discovery-benchmark-massspecgym-de-novo-molecule-generation","kind":"benchmark","links":[{"relation":"parent","target_id":"discovery-benchmark-massspecgym"},{"relation":"part_of","target_id":"discovery-benchmark-massspecgym"}],"name":"MassSpecGym De novo molecule generation","source_ids":["src-discovery-pluskal-lab-massspecgym"],"status":"discovered"} {"attributes":{"entity_level":"task","scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Rank candidate structures from a tandem mass spectrum","version":null,"historical_missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"This task ranks candidate molecular structures for an observed MS/MS spectrum.","summary_source_ids":["src-discovery-pluskal-lab-massspecgym","evidence-benchmark-massspecgym-retrieval-base-py"],"summary_source_locator":"Pinned README: three challenges; dataset and DataModule; evaluation base classes; pinned massspecgym/models/retrieval/base.py evaluation methods","sections":[{"title":"Evaluation methodology","body":"MS/MS spectrum plus a candidate set; rank the matching molecule. MCES molecular clusters are grouped into fixed training, validation and test folds, stratified by acquisition metadata. Cross-fold molecular bond-edit distance is at least 10; all spectra follow the assigned molecular fold. Mean hit rate at configured top-k cutoffs, with optional MCES distance for the top-ranked candidate. Maximum common edge subgraph (MCES) clustering keeps molecules connected by a bond-edit distance below 10 in the same fold. The split additionally balances instrument, collision-energy, adduct and molecule-frequency metadata; this is stronger than simply separating 2D InChIKeys.","source_ids":["src-discovery-pluskal-lab-massspecgym","evidence-benchmark-massspecgym-retrieval-base-py","evidence-discovery-final-massspecgym"],"source_locator":"Pinned README: three challenges; dataset and DataModule; evaluation base classes; pinned massspecgym/models/retrieval/base.py evaluation methods; Section 3.4; Supplementary Information 2.5; Section 3.4; Supplementary Information 2.5; Tables 2–4"}],"facts":[{"label":"Datasets","value":"MS/MS spectrum plus a candidate set; rank the matching molecule.","status":"source_checked","source_ids":["src-discovery-pluskal-lab-massspecgym","evidence-benchmark-massspecgym-retrieval-base-py"],"source_locator":"Pinned README: three challenges; dataset and DataModule; evaluation base classes; pinned massspecgym/models/retrieval/base.py evaluation methods"},{"label":"Splits","value":"MCES molecular clusters are grouped into fixed training, validation and test folds, stratified by acquisition metadata. Cross-fold molecular bond-edit distance is at least 10; all spectra follow the assigned molecular fold.","status":"source_checked","source_ids":["evidence-discovery-final-massspecgym"],"source_locator":"Section 3.4; Supplementary Information 2.5"},{"label":"Metrics","value":"Mean hit rate at configured top-k cutoffs, with optional MCES distance for the top-ranked candidate.","status":"source_checked","source_ids":["src-discovery-pluskal-lab-massspecgym","evidence-benchmark-massspecgym-retrieval-base-py"],"source_locator":"Pinned README: three challenges; dataset and DataModule; evaluation base classes; pinned massspecgym/models/retrieval/base.py evaluation methods"},{"label":"Baselines","value":"The README illustrates a DeepSets-style spectrum-to-fingerprint retrieval baseline.","status":"source_checked","source_ids":["src-discovery-pluskal-lab-massspecgym","evidence-benchmark-massspecgym-retrieval-base-py"],"source_locator":"Pinned README: three challenges; dataset and DataModule; evaluation base classes; pinned massspecgym/models/retrieval/base.py evaluation methods"},{"label":"Leakage controls","value":"Maximum common edge subgraph (MCES) clustering keeps molecules connected by a bond-edit distance below 10 in the same fold. The split additionally balances instrument, collision-energy, adduct and molecule-frequency metadata; this is stronger than simply separating 2D InChIKeys.","status":"source_checked","source_ids":["evidence-discovery-final-massspecgym"],"source_locator":"Section 3.4; Supplementary Information 2.5; Tables 2–4"},{"label":"Uncertainty","value":"Tables 2–4 report 99.9% bootstrap confidence intervals using 20,000 resamples. These intervals summarize test-example sampling, not variation across independently retrained models.","status":"source_checked","source_ids":["evidence-discovery-final-massspecgym"],"source_locator":"Section 3.4; Supplementary Information 2.5; Tables 2–4"},{"label":"Entity type","value":"Constituent benchmark task: MassSpecGym Molecule retrieval","status":"source_checked","source_ids":["src-discovery-pluskal-lab-massspecgym","evidence-benchmark-massspecgym-retrieval-base-py"],"source_locator":"Pinned README: three challenges; dataset and DataModule; evaluation base classes; pinned massspecgym/models/retrieval/base.py evaluation methods"},{"label":"Organisms","value":"Molecule identity rather than organism classification defines these tasks.","status":"inapplicable","source_ids":["src-discovery-pluskal-lab-massspecgym","evidence-benchmark-massspecgym-retrieval-base-py"],"source_locator":"Pinned README: three challenges; dataset and DataModule; evaluation base classes; pinned massspecgym/models/retrieval/base.py evaluation methods"},{"label":"Assays","value":"Tandem mass spectra paired with molecular structures.","status":"source_checked","source_ids":["src-discovery-pluskal-lab-massspecgym","evidence-benchmark-massspecgym-retrieval-base-py"],"source_locator":"Pinned README: three challenges; dataset and DataModule; evaluation base classes; pinned massspecgym/models/retrieval/base.py evaluation methods"},{"label":"Allowed inputs","value":"MS/MS spectrum plus the supplied molecule candidate set.","status":"source_checked","source_ids":["src-discovery-pluskal-lab-massspecgym","evidence-benchmark-massspecgym-retrieval-base-py"],"source_locator":"Pinned README: three challenges; dataset and DataModule; evaluation base classes; pinned massspecgym/models/retrieval/base.py evaluation methods"},{"label":"Adaptation","value":"Supervised train/validation/test learning; pretrained or new models use the task-specific interfaces.","status":"source_checked","source_ids":["src-discovery-pluskal-lab-massspecgym","evidence-benchmark-massspecgym-retrieval-base-py"],"source_locator":"Pinned README: three challenges; dataset and DataModule; evaluation base classes; pinned massspecgym/models/retrieval/base.py evaluation methods"}],"strengths":[{"text":"One release supplies explicit task interfaces and predefined splits for three complementary predictions.","source_ids":["src-discovery-pluskal-lab-massspecgym","evidence-benchmark-massspecgym-retrieval-base-py"],"source_locator":"Pinned README: three challenges; dataset and DataModule; evaluation base classes; pinned massspecgym/models/retrieval/base.py evaluation methods"}],"limitations":[{"text":"Chemical-formula-assisted tasks provide extra input information and must remain separate from unassisted tasks. The MCES split constrains structural similarity but cannot establish independence from every external pretraining corpus.","source_ids":["evidence-discovery-final-massspecgym"],"source_locator":"Section 3.4; Supplementary Information 2.5; Tables 2–4"}],"diagram":{"title":"Evaluation procedure","steps":["Allowed inputs: MS/MS spectrum plus the supplied molecule candidate set.","Splits: MCES molecular clusters are grouped into fixed training, validation and test folds, stratified by acquisition metadata. Cross-fold molecular bond-edit distance is at least 10; all spectra follow the assigned molecular fold.","Metrics: Mean hit rate at configured top-k cutoffs, with optional MCES distance for the top-ranked candidate."],"caption":"Conceptual procedure. Task variants and protocol versions retain their separate scoring conditions.","source_ids":["src-discovery-pluskal-lab-massspecgym","evidence-benchmark-massspecgym-retrieval-base-py","evidence-discovery-final-massspecgym"],"source_locator":"Pinned README: three challenges; dataset and DataModule; evaluation base classes; pinned massspecgym/models/retrieval/base.py evaluation methods; Section 3.4; Supplementary Information 2.5"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary paper and/or task implementation reviewed for the explicitly cited methodology claims. Scope-limited absence is recorded only after the documented source search; no model runs or independent reproduction."}}},"description":"Rank candidate structures from a tandem mass spectrum","facets":{"areas":["metabolomics"]},"id":"discovery-benchmark-massspecgym-molecule-retrieval","kind":"benchmark","links":[{"relation":"parent","target_id":"discovery-benchmark-massspecgym"},{"relation":"part_of","target_id":"discovery-benchmark-massspecgym"}],"name":"MassSpecGym Molecule retrieval","source_ids":["src-discovery-pluskal-lab-massspecgym"],"status":"discovered"} {"attributes":{"entity_level":"task","scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Predict a tandem mass spectrum from molecular structure","version":null,"historical_missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"This task predicts an MS/MS spectrum from molecular structure.","summary_source_ids":["src-discovery-pluskal-lab-massspecgym","evidence-benchmark-massspecgym-simulation-base-py"],"summary_source_locator":"Pinned README: three challenges; dataset and DataModule; evaluation base classes; pinned massspecgym/models/simulation/base.py evaluation methods","sections":[{"title":"Evaluation methodology","body":"Molecular-structure input and spectrum target; retrieval-based assessment is a distinct evaluation view. MCES molecular clusters are grouped into fixed training, validation and test folds, stratified by acquisition metadata. Cross-fold molecular bond-edit distance is at least 10; all spectra follow the assigned molecular fold. Configured cosine or Jensen–Shannon spectrum similarity, with intensity-transform variants; optional candidate-retrieval hit rates are a separate readout. Maximum common edge subgraph (MCES) clustering keeps molecules connected by a bond-edit distance below 10 in the same fold. The split additionally balances instrument, collision-energy, adduct and molecule-frequency metadata; this is stronger than simply separating 2D InChIKeys.","source_ids":["src-discovery-pluskal-lab-massspecgym","evidence-benchmark-massspecgym-simulation-base-py","evidence-discovery-final-massspecgym"],"source_locator":"Pinned README: three challenges; dataset and DataModule; evaluation base classes; pinned massspecgym/models/simulation/base.py evaluation methods; Section 3.4; Supplementary Information 2.5; Section 3.4; Supplementary Information 2.5; Tables 2–4"}],"facts":[{"label":"Datasets","value":"Molecular-structure input and spectrum target; retrieval-based assessment is a distinct evaluation view.","status":"source_checked","source_ids":["src-discovery-pluskal-lab-massspecgym","evidence-benchmark-massspecgym-simulation-base-py"],"source_locator":"Pinned README: three challenges; dataset and DataModule; evaluation base classes; pinned massspecgym/models/simulation/base.py evaluation methods"},{"label":"Splits","value":"MCES molecular clusters are grouped into fixed training, validation and test folds, stratified by acquisition metadata. Cross-fold molecular bond-edit distance is at least 10; all spectra follow the assigned molecular fold.","status":"source_checked","source_ids":["evidence-discovery-final-massspecgym"],"source_locator":"Section 3.4; Supplementary Information 2.5"},{"label":"Metrics","value":"Configured cosine or Jensen–Shannon spectrum similarity, with intensity-transform variants; optional candidate-retrieval hit rates are a separate readout.","status":"source_checked","source_ids":["src-discovery-pluskal-lab-massspecgym","evidence-benchmark-massspecgym-simulation-base-py"],"source_locator":"Pinned README: three challenges; dataset and DataModule; evaluation base classes; pinned massspecgym/models/simulation/base.py evaluation methods"},{"label":"Baselines","value":"The README illustrates a DeepSets-style spectrum-to-fingerprint retrieval baseline.","status":"source_checked","source_ids":["src-discovery-pluskal-lab-massspecgym","evidence-benchmark-massspecgym-simulation-base-py"],"source_locator":"Pinned README: three challenges; dataset and DataModule; evaluation base classes; pinned massspecgym/models/simulation/base.py evaluation methods"},{"label":"Leakage controls","value":"Maximum common edge subgraph (MCES) clustering keeps molecules connected by a bond-edit distance below 10 in the same fold. The split additionally balances instrument, collision-energy, adduct and molecule-frequency metadata; this is stronger than simply separating 2D InChIKeys.","status":"source_checked","source_ids":["evidence-discovery-final-massspecgym"],"source_locator":"Section 3.4; Supplementary Information 2.5; Tables 2–4"},{"label":"Uncertainty","value":"Tables 2–4 report 99.9% bootstrap confidence intervals using 20,000 resamples. These intervals summarize test-example sampling, not variation across independently retrained models.","status":"source_checked","source_ids":["evidence-discovery-final-massspecgym"],"source_locator":"Section 3.4; Supplementary Information 2.5; Tables 2–4"},{"label":"Entity type","value":"Constituent benchmark task: MassSpecGym Spectrum simulation","status":"source_checked","source_ids":["src-discovery-pluskal-lab-massspecgym","evidence-benchmark-massspecgym-simulation-base-py"],"source_locator":"Pinned README: three challenges; dataset and DataModule; evaluation base classes; pinned massspecgym/models/simulation/base.py evaluation methods"},{"label":"Organisms","value":"Molecule identity rather than organism classification defines these tasks.","status":"inapplicable","source_ids":["src-discovery-pluskal-lab-massspecgym","evidence-benchmark-massspecgym-simulation-base-py"],"source_locator":"Pinned README: three challenges; dataset and DataModule; evaluation base classes; pinned massspecgym/models/simulation/base.py evaluation methods"},{"label":"Assays","value":"Tandem mass spectra paired with molecular structures.","status":"source_checked","source_ids":["src-discovery-pluskal-lab-massspecgym","evidence-benchmark-massspecgym-simulation-base-py"],"source_locator":"Pinned README: three challenges; dataset and DataModule; evaluation base classes; pinned massspecgym/models/simulation/base.py evaluation methods"},{"label":"Allowed inputs","value":"Molecular structure, with the measured spectrum used only as the target.","status":"source_checked","source_ids":["src-discovery-pluskal-lab-massspecgym","evidence-benchmark-massspecgym-simulation-base-py"],"source_locator":"Pinned README: three challenges; dataset and DataModule; evaluation base classes; pinned massspecgym/models/simulation/base.py evaluation methods"},{"label":"Adaptation","value":"Supervised train/validation/test learning; pretrained or new models use the task-specific interfaces.","status":"source_checked","source_ids":["src-discovery-pluskal-lab-massspecgym","evidence-benchmark-massspecgym-simulation-base-py"],"source_locator":"Pinned README: three challenges; dataset and DataModule; evaluation base classes; pinned massspecgym/models/simulation/base.py evaluation methods"}],"strengths":[{"text":"One release supplies explicit task interfaces and predefined splits for three complementary predictions.","source_ids":["src-discovery-pluskal-lab-massspecgym","evidence-benchmark-massspecgym-simulation-base-py"],"source_locator":"Pinned README: three challenges; dataset and DataModule; evaluation base classes; pinned massspecgym/models/simulation/base.py evaluation methods"}],"limitations":[{"text":"Chemical-formula-assisted tasks provide extra input information and must remain separate from unassisted tasks. The MCES split constrains structural similarity but cannot establish independence from every external pretraining corpus.","source_ids":["evidence-discovery-final-massspecgym"],"source_locator":"Section 3.4; Supplementary Information 2.5; Tables 2–4"}],"diagram":{"title":"Evaluation procedure","steps":["Allowed inputs: Molecular structure, with the measured spectrum used only as the target.","Splits: MCES molecular clusters are grouped into fixed training, validation and test folds, stratified by acquisition metadata. Cross-fold molecular bond-edit distance is at least 10; all spectra follow the assigned molecular fold.","Metrics: Configured cosine or Jensen–Shannon spectrum similarity, with intensity-transform variants; optional candidate-retrieval hit rates are a separate readout."],"caption":"Conceptual procedure. Task variants and protocol versions retain their separate scoring conditions.","source_ids":["src-discovery-pluskal-lab-massspecgym","evidence-benchmark-massspecgym-simulation-base-py","evidence-discovery-final-massspecgym"],"source_locator":"Pinned README: three challenges; dataset and DataModule; evaluation base classes; pinned massspecgym/models/simulation/base.py evaluation methods; Section 3.4; Supplementary Information 2.5"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary paper and/or task implementation reviewed for the explicitly cited methodology claims. Scope-limited absence is recorded only after the documented source search; no model runs or independent reproduction."}}},"description":"Predict a tandem mass spectrum from molecular structure","facets":{"areas":["metabolomics"]},"id":"discovery-benchmark-massspecgym-spectrum-simulation","kind":"benchmark","links":[{"relation":"parent","target_id":"discovery-benchmark-massspecgym"},{"relation":"part_of","target_id":"discovery-benchmark-massspecgym"}],"name":"MassSpecGym Spectrum simulation","source_ids":["src-discovery-pluskal-lab-massspecgym"],"status":"discovered"} {"attributes":{"entity_level":"suite","scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"mRNA embedding quality on downstream tasks","version":null,"historical_missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"mRNABench assesses genomic-model embeddings on transcript-specific expression, stability and regulatory tasks.","summary_source_ids":["src-discovery-morrislab-mrnabench"],"summary_source_locator":"Pinned README: Overview; catalogue; dataset implementation requirements","sections":[{"title":"Evaluation methodology","body":"mRNABench evaluates mature-transcript representations on local sequence effects and global RNA properties. Linear probes use task-specific labels, with homology-based partitions where applicable. Chromosomal, k-mer and homology grouping are compared explicitly because random splits can overstate generalization.","source_ids":["evidence-discovery-final-mrnabench"],"source_locator":"Sections 3–4; Table 2; Appendix A and D"}],"facts":[{"label":"Datasets","value":"Named transcript datasets include translation efficiency, ribosome loading and RNA half-life.","status":"source_checked","source_ids":["src-discovery-morrislab-mrnabench"],"source_locator":"Pinned README: Overview; catalogue; dataset implementation requirements"},{"label":"Splits","value":"The library includes training split logic and supports homology-aware splitting using gene identifiers.","status":"source_checked","source_ids":["src-discovery-morrislab-mrnabench"],"source_locator":"Pinned README: Overview; catalogue; dataset implementation requirements"},{"label":"Metrics","value":"Task-specific AUPRC for classification and Pearson correlation for continuous RNA properties. Cross-task summaries Fisher-transform correlations before z-scoring; these derived summaries are distinct from the original per-assay metric.","status":"source_checked","source_ids":["evidence-discovery-final-mrnabench"],"source_locator":"Sections 3–4; Table 2; Appendix A and D"},{"label":"Baselines","value":"NaiveBaseline uses k-mer/GC/sequence statistics, with a six-track variant adding CDS length and exon count; NaiveMamba is an untrained fixed-seed reference.","status":"source_checked","source_ids":["src-discovery-morrislab-mrnabench"],"source_locator":"Pinned README: Overview; catalogue; dataset implementation requirements"},{"label":"Leakage controls","value":"Homology splitting requires explicit gene metadata; its presence in the library does not establish that every dataset uses it.","status":"source_checked","source_ids":["src-discovery-morrislab-mrnabench"],"source_locator":"Pinned README: Overview; catalogue; dataset implementation requirements"},{"label":"Uncertainty","value":"Linear-probe results are means over ten random splits/seeds. Appendix A and D document standard errors and the selected configurations; Table 2 also uses a Wilcoxon signed-rank comparison.","status":"source_checked","source_ids":["evidence-discovery-final-mrnabench"],"source_locator":"Sections 3–4; Table 2; Appendix A and D"},{"label":"Entity type","value":"mRNA representation benchmark suite.","status":"source_checked","source_ids":["src-discovery-morrislab-mrnabench"],"source_locator":"Pinned README: Overview; catalogue; dataset implementation requirements"},{"label":"Organisms","value":"Human and other dataset-specific transcript collections; the splitter example explicitly conditions on human homology.","status":"source_checked","source_ids":["src-discovery-morrislab-mrnabench"],"source_locator":"Pinned README: Overview; catalogue; dataset implementation requirements"},{"label":"Assays","value":"Translation efficiency, ribosome load, half-life and transcript-associated annotations.","status":"source_checked","source_ids":["src-discovery-morrislab-mrnabench"],"source_locator":"Pinned README: Overview; catalogue; dataset implementation requirements"},{"label":"Allowed inputs","value":"Transcript sequences; some feature baselines additionally use coding-region and splice annotations.","status":"source_checked","source_ids":["src-discovery-morrislab-mrnabench"],"source_locator":"Pinned README: Overview; catalogue; dataset implementation requirements"},{"label":"Adaptation","value":"Frozen embeddings with linear probes; split logic is selected independently of the embedding model.","status":"source_checked","source_ids":["src-discovery-morrislab-mrnabench"],"source_locator":"Pinned README: Overview; catalogue; dataset implementation requirements"}],"strengths":[{"text":"Naive sequence-feature and randomly initialized model baselines test whether pretraining adds useful information.","source_ids":["src-discovery-morrislab-mrnabench"],"source_locator":"Pinned README: Overview; catalogue; dataset implementation requirements"}],"limitations":[{"text":"Homology grouping is not used for every assay: the paper retains random splits for MRL-MPRA, MRL-HL-PAIR and variant effects. Derived cross-task z-scores should not replace the original biological metrics.","source_ids":["evidence-discovery-final-mrnabench"],"source_locator":"Sections 3–4; Table 2; Appendix A and D"}],"diagram":{"title":"Evaluation procedure","steps":["Allowed inputs: Transcript sequences; some feature baselines additionally use coding-region and splice annotations.","Splits: The library includes training split logic and supports homology-aware splitting using gene identifiers.","Metrics: Task-specific AUPRC for classification and Pearson correlation for continuous RNA properties. Cross-task summaries Fisher-transform correlations before z-scoring; these derived summaries are distinct from the original per-assay metric."],"caption":"Conceptual procedure. Task variants and protocol versions retain their separate scoring conditions.","source_ids":["src-discovery-morrislab-mrnabench","evidence-discovery-final-mrnabench"],"source_locator":"Pinned README: Overview; catalogue; dataset implementation requirements; Sections 3–4; Table 2; Appendix A and D"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary paper and/or task implementation reviewed for the explicitly cited methodology claims. Scope-limited absence is recorded only after the documented source search; no model runs or independent reproduction."}}},"description":"mRNA embedding quality on downstream tasks","facets":{"areas":["rna"]},"id":"discovery-benchmark-mrnabench","kind":"benchmark","links":[],"name":"mRNABench","source_ids":["src-discovery-morrislab-mrnabench"],"status":"discovered"} {"attributes":{"entity_level":"suite","scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"DNA and RNA fitness prediction","version":null,"historical_missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"NABench compares nucleotide foundation models on measured DNA/RNA sequence effects under multiple adaptation settings.","summary_source_ids":["src-discovery-mrzzmrzz-nabench"],"summary_source_locator":"Pinned README: Introduction; Evaluation settings; evaluation scripts","sections":[{"title":"Evaluation methodology","body":"NABench compares nucleic-acid fitness prediction across DMS and SELEX assays. Zero-shot sequence scores, supervised ridge probes and low-label settings are separate evaluation regimes. Random and contiguous-position folds distinguish interpolation from transfer to unseen mutational regions.","source_ids":["evidence-discovery-final-nabench"],"source_locator":"Sections on evaluation settings and metrics; dataset appendix"}],"facts":[{"label":"Datasets","value":"High-throughput assay collections spanning DNA and RNA families.","status":"source_checked","source_ids":["src-discovery-mrzzmrzz-nabench"],"source_locator":"Pinned README: Introduction; Evaluation settings; evaluation scripts"},{"label":"Splits","value":"Zero-shot, few-shot, supervised and transfer-learning settings are separate benchmark regimes.","status":"source_checked","source_ids":["src-discovery-mrzzmrzz-nabench"],"source_locator":"Pinned README: Introduction; Evaluation settings; evaluation scripts"},{"label":"Metrics","value":"Zero-shot evaluation reports Spearman correlation, NDCG, AUROC and MCC. Supervised and few-shot DMS evaluation emphasizes Spearman correlation, whereas SELEX evaluation uses AUROC.","status":"source_checked","source_ids":["evidence-discovery-final-nabench"],"source_locator":"Sections on evaluation settings and metrics; dataset appendix"},{"label":"Baselines","value":"BERT-like, GPT-like, Hyena and LLaMA-based model families are included.","status":"source_checked","source_ids":["src-discovery-mrzzmrzz-nabench"],"source_locator":"Pinned README: Introduction; Evaluation settings; evaluation scripts"},{"label":"Leakage controls","value":"Supervised probes use five-fold random or contiguous-position partitions. The contiguous version holds out variants mutated in a region of the wild-type sequence to test transfer across mutation positions; it is not a global homology or pretraining-overlap audit.","status":"source_checked","source_ids":["evidence-discovery-final-nabench"],"source_locator":"Sections on evaluation settings and metrics; dataset appendix"},{"label":"Uncertainty","value":"The inspected evaluation sections specify five-fold cross-validation and aggregate assay results, but do not define a uniform seed-based or bootstrap confidence interval for the suite.","status":"unreported","source_ids":["evidence-discovery-final-nabench"],"source_locator":"Sections on evaluation settings and metrics; dataset appendix"},{"label":"Entity type","value":"Nucleic-acid variant-effect benchmark suite.","status":"source_checked","source_ids":["src-discovery-mrzzmrzz-nabench"],"source_locator":"Pinned README: Introduction; Evaluation settings; evaluation scripts"},{"label":"Organisms","value":"Multiple natural DNA/RNA families and synthetic SELEX libraries, with assay-dependent experimental contexts. A synthetic selected sequence need not have a unique organism of origin.","status":"source_checked","source_ids":["evidence-discovery-final-nabench"],"source_locator":"Sections on evaluation settings and metrics; dataset appendix"},{"label":"Assays","value":"DMS and SELEX-derived nucleic-acid measurements; the README distinguishes released DMS data from pending processed SELEX data.","status":"source_checked","source_ids":["src-discovery-mrzzmrzz-nabench"],"source_locator":"Pinned README: Introduction; Evaluation settings; evaluation scripts"},{"label":"Allowed inputs","value":"DNA/RNA sequences and task-specific measured effects.","status":"source_checked","source_ids":["src-discovery-mrzzmrzz-nabench"],"source_locator":"Pinned README: Introduction; Evaluation settings; evaluation scripts"},{"label":"Adaptation","value":"Zero-shot, few-shot, supervised and transfer settings are evaluated separately.","status":"source_checked","source_ids":["src-discovery-mrzzmrzz-nabench"],"source_locator":"Pinned README: Introduction; Evaluation settings; evaluation scripts"}],"strengths":[{"text":"Separating label-access regimes exposes when supervised adaptation changes model comparisons.","source_ids":["src-discovery-mrzzmrzz-nabench"],"source_locator":"Pinned README: Introduction; Evaluation settings; evaluation scripts"}],"limitations":[{"text":"Fitness labels arise from different selection and reporter assays. The paper’s generalization controls do not establish that every underlying sequence was absent from model pretraining.","source_ids":["evidence-discovery-final-nabench"],"source_locator":"Sections on evaluation settings and metrics; dataset appendix"}],"diagram":{"title":"Evaluation procedure","steps":["Allowed inputs: DNA/RNA sequences and task-specific measured effects.","Splits: Zero-shot, few-shot, supervised and transfer-learning settings are separate benchmark regimes.","Metrics: Zero-shot evaluation reports Spearman correlation, NDCG, AUROC and MCC. Supervised and few-shot DMS evaluation emphasizes Spearman correlation, whereas SELEX evaluation uses AUROC."],"caption":"Conceptual procedure. Task variants and protocol versions retain their separate scoring conditions.","source_ids":["src-discovery-mrzzmrzz-nabench","evidence-discovery-final-nabench"],"source_locator":"Pinned README: Introduction; Evaluation settings; evaluation scripts; Sections on evaluation settings and metrics; dataset appendix"},"coverage":"limited","gaps":["Uncertainty: The inspected evaluation sections specify five-fold cross-validation and aggregate assay results, but do not define a uniform seed-based or bootstrap confidence interval for the suite."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary paper and/or task implementation reviewed for the explicitly cited methodology claims. Scope-limited absence is recorded only after the documented source search; no model runs or independent reproduction."}}},"description":"DNA and RNA fitness prediction","facets":{"areas":["rna"]},"id":"discovery-benchmark-nabench","kind":"benchmark","links":[],"name":"NABench","source_ids":["src-discovery-mrzzmrzz-nabench"],"status":"discovered"} {"attributes":{"entity_level":"evaluator","scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Metagenomic taxonomic profile evaluation","version":null,"historical_missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"OPAL evaluates taxon presence and abundance profiles against a reference community.","summary_source_ids":["src-discovery-cami-challenge-opal"],"summary_source_locator":"Pinned README: introduction; Computed metrics; input format; example evaluations","sections":[{"title":"Evaluation methodology","body":"Predicted taxonomic abundance profiles and a gold-standard profile in supported formats. The evaluator scores supplied outputs; predictor train/test partitions belong to the dataset/run being evaluated. Precision, recall, F1, Jaccard, L1 error, UniFrac, Bray–Curtis and diversity measures. Multiple profiling tools can be compared; published example reports use CAMI and mock-community data. Uncertainty across samples, datasets or training runs must be defined by the evaluation study; this evaluator entry does not fix one experiment.","source_ids":["src-discovery-cami-challenge-opal"],"source_locator":"Pinned README: introduction; Computed metrics; input format; example evaluations"}],"facts":[{"label":"Datasets","value":"Predicted taxonomic abundance profiles and a gold-standard profile in supported formats.","status":"source_checked","source_ids":["src-discovery-cami-challenge-opal"],"source_locator":"Pinned README: introduction; Computed metrics; input format; example evaluations"},{"label":"Splits","value":"The evaluator scores supplied outputs; predictor train/test partitions belong to the dataset/run being evaluated.","status":"inapplicable","source_ids":["src-discovery-cami-challenge-opal"],"source_locator":"Pinned README: introduction; Computed metrics; input format; example evaluations"},{"label":"Metrics","value":"Precision, recall, F1, Jaccard, L1 error, UniFrac, Bray–Curtis and diversity measures.","status":"source_checked","source_ids":["src-discovery-cami-challenge-opal"],"source_locator":"Pinned README: introduction; Computed metrics; input format; example evaluations"},{"label":"Baselines","value":"Multiple profiling tools can be compared; published example reports use CAMI and mock-community data.","status":"source_checked","source_ids":["src-discovery-cami-challenge-opal"],"source_locator":"Pinned README: introduction; Computed metrics; input format; example evaluations"},{"label":"Leakage controls","value":"OPAL scores taxonomic profiles against a supplied gold standard; it does not define or audit predictor training data. Reference-database cutoffs and novelty controls must accompany the evaluated challenge.","status":"inapplicable","source_ids":["evidence-discovery-final-opal"],"source_locator":"Implementation: input data and evaluation metrics"},{"label":"Uncertainty","value":"Uncertainty across samples, datasets or training runs must be defined by the evaluation study; this evaluator entry does not fix one experiment.","status":"inapplicable","source_ids":["src-discovery-cami-challenge-opal"],"source_locator":"Pinned README: introduction; Computed metrics; input format; example evaluations"},{"label":"Entity type","value":"Evaluator for taxonomic abundance profiling.","status":"source_checked","source_ids":["src-discovery-cami-challenge-opal"],"source_locator":"Pinned README: introduction; Computed metrics; input format; example evaluations"},{"label":"Organisms","value":"The evaluator is not restricted to a single organism; profiles describe microbial communities.","status":"inapplicable","source_ids":["src-discovery-cami-challenge-opal"],"source_locator":"Pinned README: introduction; Computed metrics; input format; example evaluations"},{"label":"Assays","value":"Taxonomic abundance profiles derived from metagenomic analyses.","status":"source_checked","source_ids":["src-discovery-cami-challenge-opal"],"source_locator":"Pinned README: introduction; Computed metrics; input format; example evaluations"},{"label":"Allowed inputs","value":"Predicted and gold-standard taxon abundances in supported formats.","status":"source_checked","source_ids":["src-discovery-cami-challenge-opal"],"source_locator":"Pinned README: introduction; Computed metrics; input format; example evaluations"},{"label":"Adaptation","value":"OPAL evaluates profiles and does not define model adaptation.","status":"inapplicable","source_ids":["src-discovery-cami-challenge-opal"],"source_locator":"Pinned README: introduction; Computed metrics; input format; example evaluations"}],"strengths":[{"text":"Separates abundance agreement from presence/absence detection.","source_ids":["src-discovery-cami-challenge-opal"],"source_locator":"Pinned README: introduction; Computed metrics; input format; example evaluations"}],"limitations":[{"text":"Taxonomic profiles must use compatible taxonomy and abundance conventions. A low profiling error does not establish independence from the reference genomes.","source_ids":["evidence-discovery-final-opal"],"source_locator":"Implementation: input data and evaluation metrics"}],"diagram":{"title":"Evaluation procedure","steps":["Allowed inputs: Predicted and gold-standard taxon abundances in supported formats.","Splits: The evaluator scores supplied outputs; predictor train/test partitions belong to the dataset/run being evaluated.","Metrics: Precision, recall, F1, Jaccard, L1 error, UniFrac, Bray–Curtis and diversity measures."],"caption":"Conceptual procedure. Task variants and protocol versions retain their separate scoring conditions.","source_ids":["src-discovery-cami-challenge-opal"],"source_locator":"Pinned README: introduction; Computed metrics; input format; example evaluations"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary paper and/or task implementation reviewed for the explicitly cited methodology claims. Scope-limited absence is recorded only after the documented source search; no model runs or independent reproduction."}}},"description":"Metagenomic taxonomic profile evaluation","facets":{"areas":["microbiome"]},"id":"discovery-benchmark-opal","kind":"benchmark","links":[],"name":"OPAL","source_ids":["src-discovery-cami-challenge-opal"],"status":"discovered"} {"attributes":{"entity_level":"suite","scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Community single-cell analysis benchmarks","version":null,"historical_missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"Open Problems is an extensible platform hosting benchmark tasks and their datasets.","summary_source_ids":["src-discovery-openproblems-bio-openproblems"],"summary_source_locator":"Pinned README: platform description and benchmark/dataset resource links","sections":[{"title":"Evaluation methodology","body":"Open Problems is a collection of separately versioned single-cell evaluation tasks. Each task defines its inputs, reference data, methods, controls and metrics. For example, label projection learns from reference labels and predicts a held-out dataset, whereas integration measures how supplied batches are combined.","source_ids":["evidence-discovery-final-openproblems-label"],"source_locator":"Label Projection v1.0.0: task description, dataset variants and controls"}],"facts":[{"label":"Datasets","value":"The README links benchmark and dataset catalogues rather than fixing one data release.","status":"source_checked","source_ids":["src-discovery-openproblems-bio-openproblems"],"source_locator":"Pinned README: platform description and benchmark/dataset resource links"},{"label":"Splits","value":"No platform-wide split exists: select a hosted benchmark task and its dataset protocol.","status":"inapplicable","source_ids":["src-discovery-openproblems-bio-openproblems"],"source_locator":"Pinned README: platform description and benchmark/dataset resource links"},{"label":"Metrics","value":"No platform-wide scientific metric applies: hosted task definitions select their own evaluator.","status":"inapplicable","source_ids":["src-discovery-openproblems-bio-openproblems"],"source_locator":"Pinned README: platform description and benchmark/dataset resource links"},{"label":"Baselines","value":"No single platform-wide baseline: methods and references belong to each hosted task.","status":"inapplicable","source_ids":["src-discovery-openproblems-bio-openproblems"],"source_locator":"Pinned README: platform description and benchmark/dataset resource links"},{"label":"Leakage controls","value":"Controls are task dependent. Label Projection v1.0.0 distinguishes training/test batches and includes both random and batch-based CeNGEN dataset variants. Other tasks, such as batch integration, evaluate transductive processing of the supplied cells; a universal held-out-cell rule would be misleading.","status":"source_checked","source_ids":["evidence-discovery-final-openproblems-label"],"source_locator":"Label Projection v1.0.0: task description, dataset variants and controls"},{"label":"Uncertainty","value":"The inspected task pages and reporting configuration do not prescribe one platform-wide bootstrap or repeated-seed interval. Individual task versions define datasets, metrics and runs; model uncertainty from a method such as scANVI is not benchmark-score uncertainty.","status":"unreported","source_ids":["evidence-discovery-final-openproblems-label"],"source_locator":"Label Projection v1.0.0: task description, dataset variants and controls"},{"label":"Entity type","value":"Platform hosting computational biology benchmark tasks.","status":"source_checked","source_ids":["src-discovery-openproblems-bio-openproblems"],"source_locator":"Pinned README: platform description and benchmark/dataset resource links"},{"label":"Organisms","value":"Organisms belong to the selected task and dataset; the platform defines no unique organism.","status":"inapplicable","source_ids":["src-discovery-openproblems-bio-openproblems"],"source_locator":"Pinned README: platform description and benchmark/dataset resource links"},{"label":"Assays","value":"The platform does not prescribe one assay.","status":"inapplicable","source_ids":["src-discovery-openproblems-bio-openproblems"],"source_locator":"Pinned README: platform description and benchmark/dataset resource links"},{"label":"Allowed inputs","value":"Inputs are specified by each hosted task.","status":"source_checked","source_ids":["src-discovery-openproblems-bio-openproblems"],"source_locator":"Pinned README: platform description and benchmark/dataset resource links"},{"label":"Adaptation","value":"Adaptation rules belong to the task implementation; the platform is not a fitted predictor.","status":"inapplicable","source_ids":["src-discovery-openproblems-bio-openproblems"],"source_locator":"Pinned README: platform description and benchmark/dataset resource links"}],"strengths":[{"text":"The platform separates task definitions from methods submitted to them.","source_ids":["src-discovery-openproblems-bio-openproblems"],"source_locator":"Pinned README: platform description and benchmark/dataset resource links"}],"limitations":[{"text":"Results from different Open Problems tasks or split variants are not interchangeable. Perfect-label controls and random-label controls calibrate particular metrics; they are not measured biological performance ceilings.","source_ids":["evidence-discovery-final-openproblems-label"],"source_locator":"Label Projection v1.0.0: task description, dataset variants and controls"}],"diagram":{"title":"Evaluation procedure","steps":["Allowed inputs: Inputs are specified by each hosted task.","Splits: No platform-wide split exists: select a hosted benchmark task and its dataset protocol.","Metrics: No platform-wide scientific metric applies: hosted task definitions select their own evaluator."],"caption":"Conceptual procedure. Task variants and protocol versions retain their separate scoring conditions.","source_ids":["src-discovery-openproblems-bio-openproblems"],"source_locator":"Pinned README: platform description and benchmark/dataset resource links"},"coverage":"limited","gaps":["Uncertainty: The inspected task pages and reporting configuration do not prescribe one platform-wide bootstrap or repeated-seed interval. Individual task versions define datasets, metrics and runs; model uncertainty from a method such as scANVI is not benchmark-score uncertainty."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary paper and/or task implementation reviewed for the explicitly cited methodology claims. Scope-limited absence is recorded only after the documented source search; no model runs or independent reproduction."}}},"description":"Community single-cell analysis benchmarks","facets":{"areas":["single-cell"]},"id":"discovery-benchmark-open-problems","kind":"benchmark","links":[{"relation":"evaluates_task","target_id":"catalog-task-cell-batch-integration"}],"name":"Open Problems","source_ids":["src-discovery-openproblems-bio-openproblems"],"status":"discovered"} {"attributes":{"entity_level":"suite","scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Predicting cellular perturbation responses","version":null,"historical_missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"PerturBench evaluates predicted single-cell perturbation responses with explicit aggregation and metric choices.","summary_source_ids":["src-discovery-altoslabs-perturbench"],"summary_source_locator":"Pinned README: Data; Model evaluation; custom dataset configuration","sections":[{"title":"Evaluation methodology","body":"PerturBench predicts single-cell responses in held-out perturbation–context combinations. It implements cross-covariate, combinatorial and inverse-combinatorial partitions, comparing learned models with simple controls. Rank-based metrics complement expression-error metrics to reveal models that fail to distinguish perturbations.","source_ids":["evidence-discovery-final-perturbench"],"source_locator":"Experimental setup; Appendix datasets and data splitting"}],"facts":[{"label":"Datasets","value":"Processed AnnData datasets with perturbation/covariate metadata and configurable feature selections.","status":"source_checked","source_ids":["src-discovery-altoslabs-perturbench"],"source_locator":"Pinned README: Data; Model evaluation; custom dataset configuration"},{"label":"Splits","value":"Cross-cell-type and combination-prediction splits are supported, along with explicit custom split files.","status":"source_checked","source_ids":["src-discovery-altoslabs-perturbench"],"source_locator":"Pinned README: Data; Model evaluation; custom dataset configuration"},{"label":"Metrics","value":"Expression/change aggregation precedes metrics such as cosine, Pearson, RMSE, MSE, MAE and R-squared; optional rank metrics form another view.","status":"source_checked","source_ids":["src-discovery-altoslabs-perturbench"],"source_locator":"Pinned README: Data; Model evaluation; custom dataset configuration"},{"label":"Baselines","value":"Reproduction configurations include linear reference models.","status":"source_checked","source_ids":["src-discovery-altoslabs-perturbench"],"source_locator":"Pinned README: Data; Model evaluation; custom dataset configuration"},{"label":"Leakage controls","value":"Split and covariate definitions are configuration inputs; exact evaluation files must be pinned.","status":"source_checked","source_ids":["src-discovery-altoslabs-perturbench"],"source_locator":"Pinned README: Data; Model evaluation; custom dataset configuration"},{"label":"Uncertainty","value":"For the best hyperparameter configuration the authors run four additional training seeds, yielding five runs. Error bars represent standard deviation of model performance across those runs.","status":"source_checked","source_ids":["evidence-discovery-final-perturbench"],"source_locator":"Experimental setup; Appendix datasets and data splitting"},{"label":"Entity type","value":"Perturbation-response evaluation framework.","status":"source_checked","source_ids":["src-discovery-altoslabs-perturbench"],"source_locator":"Pinned README: Data; Model evaluation; custom dataset configuration"},{"label":"Organisms","value":"The evaluated datasets use human cell-line perturbation systems, including McFaline-Figueroa’s glioblastoma cell contexts and Srivatsan’s chemical perturbation cell lines. The framework itself is not restricted to those organisms.","status":"source_checked","source_ids":["evidence-discovery-final-perturbench"],"source_locator":"Experimental setup; Appendix datasets and data splitting"},{"label":"Assays","value":"Single-cell perturbation response measurements.","status":"source_checked","source_ids":["src-discovery-altoslabs-perturbench"],"source_locator":"Pinned README: Data; Model evaluation; custom dataset configuration"},{"label":"Allowed inputs","value":"Predicted/observed expression and perturbation/covariate metadata in AnnData.","status":"source_checked","source_ids":["src-discovery-altoslabs-perturbench"],"source_locator":"Pinned README: Data; Model evaluation; custom dataset configuration"},{"label":"Adaptation","value":"Supports supervised response prediction with explicit cell-type and combination holdouts.","status":"source_checked","source_ids":["src-discovery-altoslabs-perturbench"],"source_locator":"Pinned README: Data; Model evaluation; custom dataset configuration"}],"strengths":[{"text":"Separating response aggregation from scoring makes a major source of metric disagreement explicit.","source_ids":["src-discovery-altoslabs-perturbench"],"source_locator":"Pinned README: Data; Model evaluation; custom dataset configuration"}],"limitations":[{"text":"A covariate-transfer split can expose a perturbation in another cell context during training; it should not be called completely unseen-perturbation prediction. Report the exact split and perturbation inputs.","source_ids":["evidence-discovery-final-perturbench"],"source_locator":"Experimental setup; Appendix datasets and data splitting"}],"diagram":{"title":"Evaluation procedure","steps":["Allowed inputs: Predicted/observed expression and perturbation/covariate metadata in AnnData.","Splits: Cross-cell-type and combination-prediction splits are supported, along with explicit custom split files.","Metrics: Expression/change aggregation precedes metrics such as cosine, Pearson, RMSE, MSE, MAE and R-squared; optional rank metrics form another view."],"caption":"Conceptual procedure. Task variants and protocol versions retain their separate scoring conditions.","source_ids":["src-discovery-altoslabs-perturbench"],"source_locator":"Pinned README: Data; Model evaluation; custom dataset configuration"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary paper and/or task implementation reviewed for the explicitly cited methodology claims. Scope-limited absence is recorded only after the documented source search; no model runs or independent reproduction."}}},"description":"Predicting cellular perturbation responses","facets":{"areas":["single-cell"]},"id":"discovery-benchmark-perturbench","kind":"benchmark","links":[],"name":"PerturBench","source_ids":["src-discovery-altoslabs-perturbench"],"status":"discovered"} {"attributes":{"entity_level":"suite","scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Parameter estimation for biological dynamical models","version":null,"historical_missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"The PEtab collection supports evaluation of computational methods for fitting mathematical models to observations.","summary_source_ids":["src-discovery-benchmarking-initiative-benchmark-models-petab"],"summary_source_locator":"Pinned README: collection description; benchmark problem table","sections":[{"title":"Evaluation methodology","body":"The collection packages dynamical biological models with the experimental measurements and observation/noise assumptions needed for parameter estimation. A benchmark run specifies a model, solver and inference procedure against this fixed problem. Calibration fit, computational reliability and predictive validation are different assessment targets.","source_ids":["evidence-discovery-final-petab"],"source_locator":"Sections 2.2–2.4: observations, noise models and experimental conditions"}],"facts":[{"label":"Datasets","value":"Individual model/measurement problems in PEtab format, with problem-specific observables and noise assumptions.","status":"source_checked","source_ids":["src-discovery-benchmarking-initiative-benchmark-models-petab"],"source_locator":"Pinned README: collection description; benchmark problem table"},{"label":"Splits","value":"A supervised train/test split is not intrinsic to the parameter-estimation problem collection; the chosen study must define any held-out observations.","status":"inapplicable","source_ids":["src-discovery-benchmarking-initiative-benchmark-models-petab"],"source_locator":"Pinned README: collection description; benchmark problem table"},{"label":"Metrics","value":"Objectives depend on each PEtab measurement/noise model and the selected optimization-performance criterion; there is no universal prediction metric.","status":"inapplicable","source_ids":["src-discovery-benchmarking-initiative-benchmark-models-petab"],"source_locator":"Pinned README: collection description; benchmark problem table"},{"label":"Baselines","value":"The collection provides common problem definitions for comparing modeling/estimation methods, not a fixed universal baseline.","status":"source_checked","source_ids":["src-discovery-benchmarking-initiative-benchmark-models-petab"],"source_locator":"Pinned README: collection description; benchmark problem table"},{"label":"Leakage controls","value":"The collection provides mechanistic models with calibration data, observation functions and experimental conditions. It is intended to compare numerical inference methods; a predictive train/test exclusion policy must be defined by the study using each model.","status":"inapplicable","source_ids":["evidence-discovery-final-petab"],"source_locator":"Sections 2.2–2.4: observations, noise models and experimental conditions"},{"label":"Uncertainty","value":"Measurement errors can be fixed from experiments or estimated jointly through explicit noise models. Parameter uncertainty and optimizer variability are different quantities and require their own evaluation procedure.","status":"source_checked","source_ids":["evidence-discovery-final-petab"],"source_locator":"Sections 2.2–2.4: observations, noise models and experimental conditions"},{"label":"Entity type","value":"Collection of parameter-estimation benchmark problems.","status":"source_checked","source_ids":["src-discovery-benchmarking-initiative-benchmark-models-petab"],"source_locator":"Pinned README: collection description; benchmark problem table"},{"label":"Organisms","value":"Organism identity is problem-specific; the collection spans distinct systems.","status":"inapplicable","source_ids":["src-discovery-benchmarking-initiative-benchmark-models-petab"],"source_locator":"Pinned README: collection description; benchmark problem table"},{"label":"Assays","value":"Problem-specific observations with explicit measurement/noise models.","status":"source_checked","source_ids":["src-discovery-benchmarking-initiative-benchmark-models-petab"],"source_locator":"Pinned README: collection description; benchmark problem table"},{"label":"Allowed inputs","value":"PEtab model, parameter, condition, observable and measurement tables.","status":"source_checked","source_ids":["src-discovery-benchmarking-initiative-benchmark-models-petab"],"source_locator":"Pinned README: collection description; benchmark problem table"},{"label":"Adaptation","value":"Numerical parameter estimation against supplied observations; optimizer settings define the tested method.","status":"source_checked","source_ids":["src-discovery-benchmarking-initiative-benchmark-models-petab"],"source_locator":"Pinned README: collection description; benchmark problem table"}],"strengths":[{"text":"Standardized problem definitions expose differences in objectives and measurement-error assumptions.","source_ids":["src-discovery-benchmarking-initiative-benchmark-models-petab"],"source_locator":"Pinned README: collection description; benchmark problem table"}],"limitations":[{"text":"The original collection paper describes the scientific problem format; the current PEtab repository is a later versioned distribution. Fitting calibration data is not evidence of accuracy on an independent experimental condition.","source_ids":["evidence-discovery-final-petab"],"source_locator":"Sections 2.2–2.4: observations, noise models and experimental conditions"}],"diagram":{"title":"Evaluation procedure","steps":["Allowed inputs: PEtab model, parameter, condition, observable and measurement tables.","Splits: A supervised train/test split is not intrinsic to the parameter-estimation problem collection; the chosen study must define any held-out observations.","Metrics: Objectives depend on each PEtab measurement/noise model and the selected optimization-performance criterion; there is no universal prediction metric."],"caption":"Conceptual procedure. Task variants and protocol versions retain their separate scoring conditions.","source_ids":["src-discovery-benchmarking-initiative-benchmark-models-petab"],"source_locator":"Pinned README: collection description; benchmark problem table"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary paper and/or task implementation reviewed for the explicitly cited methodology claims. Scope-limited absence is recorded only after the documented source search; no model runs or independent reproduction."}}},"description":"Parameter estimation for biological dynamical models","facets":{"areas":["mechanistic-biology"]},"id":"discovery-benchmark-petab-benchmark-collection","kind":"benchmark","links":[],"name":"PEtab benchmark collection","source_ids":["src-discovery-benchmarking-initiative-benchmark-models-petab"],"status":"discovered"} {"attributes":{"entity_level":"suite","scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Protein representation evaluation","version":null,"historical_missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"PFMBench is a configurable suite of protein-model downstream evaluations.","summary_source_ids":["src-discovery-biomap-research-pfmbench"],"summary_source_locator":"Pinned README: Overview; Features; repository architecture","sections":[{"title":"Evaluation methodology","body":"PFMBench evaluates protein representations across structural, functional, interaction and engineering tasks. Most datasets use sequence-similarity partitions, while mutation datasets retain their original assay splits. A stability screen identifies a core task subset, which must remain distinguishable from the full collection.","source_ids":["evidence-discovery-final-pfmbench"],"source_locator":"Benchmark construction and evaluation setup; Appendix task definitions"}],"facts":[{"label":"Datasets","value":"Tasks span structure, function, localization, interactions and other protein properties.","status":"source_checked","source_ids":["src-discovery-biomap-research-pfmbench"],"source_locator":"Pinned README: Overview; Features; repository architecture"},{"label":"Splits","value":"Most datasets are split 8:1:1 using a 30% protein sequence-similarity threshold. Mutation datasets are explicitly exempt and preserve their original train/validation/test partitions.","status":"source_checked","source_ids":["evidence-discovery-final-pfmbench"],"source_locator":"Benchmark construction"},{"label":"Metrics","value":"Task-specific metrics include AUROC for several binary function and interaction tasks, accuracy for categorical tasks, and Spearman correlation for continuous fitness, affinity and enzyme properties; the task appendix defines the metric for each dataset.","status":"source_checked","source_ids":["evidence-discovery-final-pfmbench"],"source_locator":"Benchmark construction and evaluation setup; Appendix task definitions"},{"label":"Baselines","value":"The framework supports both fine-tuning on labels and zero-shot evaluations.","status":"source_checked","source_ids":["src-discovery-biomap-research-pfmbench"],"source_locator":"Pinned README: Overview; Features; repository architecture"},{"label":"Leakage controls","value":"Most datasets use an 8:1:1 split with a 30% sequence-similarity threshold. Mutation datasets retain their original partitions. The paper also flags possible functional-label overlap for annotation-aware pretrained models.","status":"source_checked","source_ids":["evidence-discovery-final-pfmbench"],"source_locator":"Benchmark construction and evaluation setup; Appendix task definitions"},{"label":"Uncertainty","value":"ESM2-Adapter is evaluated over three runs to screen task stability. Its reported bias is the best-to-worst spread divided by mean performance; this is not a confidence interval or a rule proven for all models.","status":"source_checked","source_ids":["evidence-discovery-final-pfmbench"],"source_locator":"Benchmark construction and evaluation setup; Appendix task definitions"},{"label":"Entity type","value":"Configurable protein foundation-model evaluation suite.","status":"source_checked","source_ids":["src-discovery-biomap-research-pfmbench"],"source_locator":"Pinned README: Overview; Features; repository architecture"},{"label":"Organisms","value":"Dataset dependent: named tasks include human and yeast protein interactions, broader protein-property collections, molecular binding data and mutation assays. Species is a property of each source dataset, not one suite-wide organism.","status":"source_checked","source_ids":["evidence-discovery-final-pfmbench"],"source_locator":"Benchmark construction and evaluation setup; Appendix task definitions"},{"label":"Assays","value":"Task-specific structure, function, localization and interaction labels.","status":"source_checked","source_ids":["src-discovery-biomap-research-pfmbench"],"source_locator":"Pinned README: Overview; Features; repository architecture"},{"label":"Allowed inputs","value":"Protein task datasets via configurable loaders and prediction heads.","status":"source_checked","source_ids":["src-discovery-biomap-research-pfmbench"],"source_locator":"Pinned README: Overview; Features; repository architecture"},{"label":"Adaptation","value":"Both fine-tuning and zero-shot evaluation are supported; configurations define the adaptation budget.","status":"source_checked","source_ids":["src-discovery-biomap-research-pfmbench"],"source_locator":"Pinned README: Overview; Features; repository architecture"}],"strengths":[{"text":"The same task framework supports multiple representation and tuning strategies.","source_ids":["src-discovery-biomap-research-pfmbench"],"source_locator":"Pinned README: Overview; Features; repository architecture"}],"limitations":[{"text":"Core-task selection based on one adapter’s repeated runs does not guarantee equal reliability for every model. Functional-label pretraining and mutation-specific splits require separate overlap checks.","source_ids":["evidence-discovery-final-pfmbench"],"source_locator":"Benchmark construction and evaluation setup; Appendix task definitions"}],"diagram":{"title":"Evaluation procedure","steps":["Allowed inputs: Protein task datasets via configurable loaders and prediction heads.","Splits: Most datasets are split 8:1:1 using a 30% protein sequence-similarity threshold. Mutation datasets are explicitly exempt and preserve their original train/validation/test partitions.","Metrics: Task-specific metrics include AUROC for several binary function and interaction tasks, accuracy for categorical tasks, and Spearman correlation for continuous fitness, affinity and enzyme properties; the task appendix defines the metric for each dataset."],"caption":"Conceptual procedure. Task variants and protocol versions retain their separate scoring conditions.","source_ids":["src-discovery-biomap-research-pfmbench","evidence-discovery-final-pfmbench"],"source_locator":"Pinned README: Overview; Features; repository architecture; Benchmark construction; Benchmark construction and evaluation setup; Appendix task definitions"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary paper and/or task implementation reviewed for the explicitly cited methodology claims. Scope-limited absence is recorded only after the documented source search; no model runs or independent reproduction."}}},"description":"Protein representation evaluation","facets":{"areas":["protein-function"]},"id":"discovery-benchmark-pfmbench","kind":"benchmark","links":[],"name":"PFMBench","source_ids":["src-discovery-biomap-research-pfmbench"],"status":"discovered"} {"attributes":{"entity_level":"suite","scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Protein-ligand interaction evaluation","version":null,"historical_missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"PLINDER supplies annotated protein–ligand systems and evaluation resources for docking.","summary_source_ids":["src-discovery-plinder-org-plinder"],"summary_source_locator":"Pinned README: Overview; Plinder versions; Known bugs","sections":[{"title":"Evaluation methodology","body":"PLINDER organizes protein–ligand complexes and their similarity relationships so a test set can be characterized by novelty of proteins, pockets, ligands and interactions. The evaluator compares predicted poses with reference systems and preserves matched-chain coverage. Dataset version, split and novelty stratum are essential parts of any reported result.","source_ids":["evidence-discovery-final-plinder-config0"],"source_locator":"Pinned docs/evaluation.md: per-pose scores and test stratification"}],"facts":[{"label":"Datasets","value":"Protein–ligand complexes with linked bound, unbound and predicted receptor structures.","status":"source_checked","source_ids":["src-discovery-plinder-org-plinder"],"source_locator":"Pinned README: Overview; Plinder versions; Known bugs"},{"label":"Splits","value":"Train/validation/test splits can be tuned to the learning task.","status":"source_checked","source_ids":["src-discovery-plinder-org-plinder"],"source_locator":"Pinned README: Overview; Plinder versions; Known bugs"},{"label":"Metrics","value":"Pose scoring includes ligand lDDT-PLI, binding-site-superposed symmetry-corrected RMSD and pocket lDDT. System summaries retain mapped-chain fractions and optionally score receptor lDDT, oligomer interfaces and PoseBusters validity. A pose confidence score is optional input, not benchmark uncertainty.","status":"source_checked","source_ids":["evidence-discovery-final-plinder-config0"],"source_locator":"Pinned docs/evaluation.md: Write scores"},{"label":"Baselines","value":"The official release history identifies a dataset version used to retrain DiffDock and points to the companion Moving Beyond Memorization study. Baseline identity must include the particular PLINDER release/split; a dataset entry does not define one universal reference score.","status":"source_checked","source_ids":["evidence-discovery-final-plinder-readme"],"source_locator":"README: dataset versions and Moving Beyond Memorization reference"},{"label":"Leakage controls","value":"Similarity annotations support task-dependent splitting; a specific split is required before claiming overlap exclusion.","status":"source_checked","source_ids":["src-discovery-plinder-org-plinder"],"source_locator":"Pinned README: Overview; Plinder versions; Known bugs"},{"label":"Uncertainty","value":"The inspected evaluator documentation defines per-pose metrics, system averages and similarity strata but no universal bootstrap or repeated-training interval. An uncertainty estimate must be attached to a specific evaluated model and dataset release.","status":"unreported","source_ids":["evidence-discovery-final-plinder-config0"],"source_locator":"Pinned docs/evaluation.md: per-pose scores and test stratification"},{"label":"Entity type","value":"Protein–ligand dataset and task-dependent split resource.","status":"source_checked","source_ids":["src-discovery-plinder-org-plinder"],"source_locator":"Pinned README: Overview; Plinder versions; Known bugs"},{"label":"Organisms","value":"Molecular systems define the collection; it is not tied to one organism.","status":"inapplicable","source_ids":["src-discovery-plinder-org-plinder"],"source_locator":"Pinned README: Overview; Plinder versions; Known bugs"},{"label":"Assays","value":"Protein–ligand structural data with bound, unbound and predicted receptor states.","status":"source_checked","source_ids":["src-discovery-plinder-org-plinder"],"source_locator":"Pinned README: Overview; Plinder versions; Known bugs"},{"label":"Allowed inputs","value":"Protein–ligand systems, receptor structures and curated metadata.","status":"source_checked","source_ids":["src-discovery-plinder-org-plinder"],"source_locator":"Pinned README: Overview; Plinder versions; Known bugs"},{"label":"Adaptation","value":"The chosen downstream model and split determine fitting; the data resource imposes no single adaptation scheme.","status":"source_checked","source_ids":["src-discovery-plinder-org-plinder"],"source_locator":"Pinned README: Overview; Plinder versions; Known bugs"}],"strengths":[{"text":"Similarity annotations make protein/ligand overlap inspectable when choosing splits.","source_ids":["src-discovery-plinder-org-plinder"],"source_locator":"Pinned README: Overview; Plinder versions; Known bugs"}],"limitations":[{"text":"A split can be novel by one molecular similarity measure and familiar by another. The documented evaluator produces scores and strata; it does not certify every model’s training history or impose a common uncertainty protocol.","source_ids":["evidence-discovery-final-plinder-config0"],"source_locator":"Pinned docs/evaluation.md: per-pose scores and test stratification"}],"diagram":{"title":"Evaluation procedure","steps":["Allowed inputs: Protein–ligand systems, receptor structures and curated metadata.","Splits: Train/validation/test splits can be tuned to the learning task.","Metrics: Pose scoring includes ligand lDDT-PLI, binding-site-superposed symmetry-corrected RMSD and pocket lDDT. System summaries retain mapped-chain fractions and optionally score receptor lDDT, oligomer interfaces and PoseBusters validity. A pose confidence score is optional input, not benchmark uncertainty."],"caption":"Conceptual procedure. Task variants and protocol versions retain their separate scoring conditions.","source_ids":["src-discovery-plinder-org-plinder","evidence-discovery-final-plinder-config0"],"source_locator":"Pinned README: Overview; Plinder versions; Known bugs; Pinned docs/evaluation.md: Write scores"},"coverage":"limited","gaps":["Uncertainty: The inspected evaluator documentation defines per-pose metrics, system averages and similarity strata but no universal bootstrap or repeated-training interval. An uncertainty estimate must be attached to a specific evaluated model and dataset release."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary paper and/or task implementation reviewed for the explicitly cited methodology claims. Scope-limited absence is recorded only after the documented source search; no model runs or independent reproduction."}}},"description":"Protein-ligand interaction evaluation","facets":{"areas":["molecular-interactions"]},"id":"discovery-benchmark-plinder","kind":"benchmark","links":[],"name":"PLINDER","source_ids":["src-discovery-plinder-org-plinder"],"status":"discovered"} {"attributes":{"entity_level":"evaluator","scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Geometric and chemical plausibility of molecular poses","version":null,"historical_missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"PoseBusters checks the plausibility of predicted molecular poses.","summary_source_ids":["src-discovery-maabuu-posebusters"],"summary_source_locator":"Pinned README: description; Usage; paper/data links","sections":[{"title":"Evaluation methodology","body":"Predicted molecular coordinates, optionally with the conditioning protein, and paper-linked evaluation data. The evaluator scores supplied outputs; predictor train/test partitions belong to the dataset/run being evaluated. Chemical identity/stereochemistry, bond and angle geometry, aromatic planarity, internal clashes and protein–ligand clashes are checked separately. The paper evaluates native-like pose recovery jointly with passing the validity checks; its intramolecular tolerances are 25% for bond lengths/angles and 30% for nonbonded distances. The PoseBusters benchmark selects recent PDB complexes absent from the PDBbind v2020 training source used by evaluated learned docking methods. The paper further examines protein-sequence similarity to training data; new deposition date alone does not imply remote homology.","source_ids":["src-discovery-maabuu-posebusters","evidence-discovery-final-posebusters"],"source_locator":"Pinned README: description; Usage; paper/data links; Methods: chemical, intramolecular and intermolecular validity; Methods: benchmark construction and evaluation of generalization"}],"facts":[{"label":"Datasets","value":"Predicted molecular coordinates, optionally with the conditioning protein, and paper-linked evaluation data.","status":"source_checked","source_ids":["src-discovery-maabuu-posebusters"],"source_locator":"Pinned README: description; Usage; paper/data links"},{"label":"Splits","value":"The evaluator scores supplied outputs; predictor train/test partitions belong to the dataset/run being evaluated.","status":"inapplicable","source_ids":["src-discovery-maabuu-posebusters"],"source_locator":"Pinned README: description; Usage; paper/data links"},{"label":"Metrics","value":"Chemical identity/stereochemistry, bond and angle geometry, aromatic planarity, internal clashes and protein–ligand clashes are checked separately. The paper evaluates native-like pose recovery jointly with passing the validity checks; its intramolecular tolerances are 25% for bond lengths/angles and 30% for nonbonded distances.","status":"source_checked","source_ids":["evidence-discovery-final-posebusters"],"source_locator":"Methods: chemical, intramolecular and intermolecular validity"},{"label":"Baselines","value":"The evaluator does not prescribe a universal baseline predictor; comparisons require methods run on the same selected dataset.","status":"inapplicable","source_ids":["src-discovery-maabuu-posebusters"],"source_locator":"Pinned README: description; Usage; paper/data links"},{"label":"Leakage controls","value":"The PoseBusters benchmark selects recent PDB complexes absent from the PDBbind v2020 training source used by evaluated learned docking methods. The paper further examines protein-sequence similarity to training data; new deposition date alone does not imply remote homology.","status":"source_checked","source_ids":["evidence-discovery-final-posebusters"],"source_locator":"Methods: benchmark construction and evaluation of generalization"},{"label":"Uncertainty","value":"Uncertainty across samples, datasets or training runs must be defined by the evaluation study; this evaluator entry does not fix one experiment.","status":"inapplicable","source_ids":["src-discovery-maabuu-posebusters"],"source_locator":"Pinned README: description; Usage; paper/data links"},{"label":"Entity type","value":"Molecular-pose plausibility evaluator.","status":"source_checked","source_ids":["src-discovery-maabuu-posebusters"],"source_locator":"Pinned README: description; Usage; paper/data links"},{"label":"Organisms","value":"Plausibility checks concern molecular geometry rather than organism identity.","status":"inapplicable","source_ids":["src-discovery-maabuu-posebusters"],"source_locator":"Pinned README: description; Usage; paper/data links"},{"label":"Assays","value":"Predicted poses assessed against molecular validity criteria.","status":"source_checked","source_ids":["src-discovery-maabuu-posebusters"],"source_locator":"Pinned README: description; Usage; paper/data links"},{"label":"Allowed inputs","value":"Predicted molecule coordinates, optionally paired with a protein structure.","status":"source_checked","source_ids":["src-discovery-maabuu-posebusters"],"source_locator":"Pinned README: description; Usage; paper/data links"},{"label":"Adaptation","value":"The checker evaluates poses; it does not train or fine-tune the pose predictor.","status":"inapplicable","source_ids":["src-discovery-maabuu-posebusters"],"source_locator":"Pinned README: description; Usage; paper/data links"}],"strengths":[{"text":"Checks plausibility separately from pose agreement, preventing the two criteria being conflated.","source_ids":["src-discovery-maabuu-posebusters"],"source_locator":"Pinned README: description; Usage; paper/data links"}],"limitations":[{"text":"PoseBusters checks physical plausibility and geometric agreement. Passing those checks does not establish binding affinity, biological activity or complete independence from every external training corpus.","source_ids":["evidence-discovery-final-posebusters"],"source_locator":"Methods: benchmark construction and evaluation of generalization"}],"diagram":{"title":"Evaluation procedure","steps":["Allowed inputs: Predicted molecule coordinates, optionally paired with a protein structure.","Splits: The evaluator scores supplied outputs; predictor train/test partitions belong to the dataset/run being evaluated.","Metrics: Chemical identity/stereochemistry, bond and angle geometry, aromatic planarity, internal clashes and protein–ligand clashes are checked separately. The paper evaluates native-like pose recovery jointly with passing the validity checks; its intramolecular tolerances are 25% for bond lengths/angles and 30% for nonbonded distances."],"caption":"Conceptual procedure. Task variants and protocol versions retain their separate scoring conditions.","source_ids":["src-discovery-maabuu-posebusters","evidence-discovery-final-posebusters"],"source_locator":"Pinned README: description; Usage; paper/data links; Methods: chemical, intramolecular and intermolecular validity"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary paper and/or task implementation reviewed for the explicitly cited methodology claims. Scope-limited absence is recorded only after the documented source search; no model runs or independent reproduction."}}},"description":"Geometric and chemical plausibility of molecular poses","facets":{"areas":["molecular-interactions"]},"id":"discovery-benchmark-posebusters","kind":"benchmark","links":[],"name":"PoseBusters","source_ids":["src-discovery-maabuu-posebusters"],"status":"discovered"} {"attributes":{"entity_level":"suite","scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Protein prediction, design and dynamics evaluation","version":null,"historical_missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"ProteinBench assesses multiple protein-model tasks using quality, novelty, diversity and robustness dimensions.","summary_source_ids":["evidence-benchmark-proteinbench-snapshot"],"summary_source_locator":"ProteinBench official website: Abstract; sequence/structure task evaluations; metric descriptions","sections":[{"title":"Evaluation methodology","body":"ProteinBench evaluates protein sequence and structure methods across design, folding and conformational tasks. It distinguishes quality, diversity and novelty instead of treating all protein capabilities as one accuracy score. Each task has its own reference data, sampling procedure and baseline family.","source_ids":["evidence-discovery-final-proteinbench"],"source_locator":"Task definitions; ATLAS evaluation; antibody-design setup; result tables"}],"facts":[{"label":"Datasets","value":"Task-specific structure, sequence and complex evaluation collections.","status":"source_checked","source_ids":["evidence-benchmark-proteinbench-snapshot"],"source_locator":"ProteinBench official website: Abstract; sequence/structure task evaluations; metric descriptions"},{"label":"Splits","value":"The framework distinguishes natural in-distribution structures from generated-backbone out-of-distribution evaluations.","status":"source_checked","source_ids":["evidence-benchmark-proteinbench-snapshot"],"source_locator":"ProteinBench official website: Abstract; sequence/structure task evaluations; metric descriptions"},{"label":"Metrics","value":"Metric sets vary by task and include structure-predictor confidence, structural similarity and diversity measures.","status":"source_checked","source_ids":["evidence-benchmark-proteinbench-snapshot"],"source_locator":"ProteinBench official website: Abstract; sequence/structure task evaluations; metric descriptions"},{"label":"Baselines","value":"Comparators vary by task: structure predictors, inverse-folding/design models, Rosetta-based antibody methods and molecular-dynamics or ensemble references. The paper defines the eligible method and input information separately for each task.","status":"source_checked","source_ids":["evidence-discovery-final-proteinbench"],"source_locator":"Task definitions; ATLAS evaluation; antibody-design setup; result tables"},{"label":"Leakage controls","value":"The antibody setup clusters CDR-H3 sequences at 40% similarity and excludes clusters containing RAbD test complexes from training/validation. The ATLAS ensemble task also applies a held-out protocol for models trained on ATLAS; these are task-specific controls, not a universal suite split.","status":"source_checked","source_ids":["evidence-discovery-final-proteinbench"],"source_locator":"Task definitions; ATLAS evaluation; antibody-design setup; result tables"},{"label":"Uncertainty","value":"Some task tables report repeated-experiment averages and standard deviations; others report medians.","status":"source_checked","source_ids":["evidence-benchmark-proteinbench-snapshot"],"source_locator":"ProteinBench official website: Abstract; sequence/structure task evaluations; metric descriptions"},{"label":"Entity type","value":"Protein-model evaluation framework spanning several tasks.","status":"source_checked","source_ids":["evidence-benchmark-proteinbench-snapshot"],"source_locator":"ProteinBench official website: Abstract; sequence/structure task evaluations; metric descriptions"},{"label":"Organisms","value":"No single organism defines its sequence, structure and complex task collections.","status":"inapplicable","source_ids":["evidence-benchmark-proteinbench-snapshot"],"source_locator":"ProteinBench official website: Abstract; sequence/structure task evaluations; metric descriptions"},{"label":"Assays","value":"Task-dependent protein structure and sequence/property references.","status":"source_checked","source_ids":["evidence-benchmark-proteinbench-snapshot"],"source_locator":"ProteinBench official website: Abstract; sequence/structure task evaluations; metric descriptions"},{"label":"Allowed inputs","value":"Task-specific sequences, structures or complexes.","status":"source_checked","source_ids":["evidence-benchmark-proteinbench-snapshot"],"source_locator":"ProteinBench official website: Abstract; sequence/structure task evaluations; metric descriptions"},{"label":"Adaptation","value":"Tasks define their own generation, prediction or adaptation regimes.","status":"source_checked","source_ids":["evidence-benchmark-proteinbench-snapshot"],"source_locator":"ProteinBench official website: Abstract; sequence/structure task evaluations; metric descriptions"}],"strengths":[{"text":"Separate task collections expose which capability a result measures.","source_ids":["evidence-benchmark-proteinbench-snapshot"],"source_locator":"ProteinBench official website: Abstract; sequence/structure task evaluations; metric descriptions"}],"limitations":[{"text":"Architectures receive different inputs and can generate different numbers of candidates. Task-specific exclusion rules and sampling budgets must accompany any comparison.","source_ids":["evidence-discovery-final-proteinbench"],"source_locator":"Task definitions; ATLAS evaluation; antibody-design setup; result tables"}],"diagram":{"title":"Evaluation procedure","steps":["Allowed inputs: Task-specific sequences, structures or complexes.","Splits: The framework distinguishes natural in-distribution structures from generated-backbone out-of-distribution evaluations.","Metrics: Metric sets vary by task and include structure-predictor confidence, structural similarity and diversity measures."],"caption":"Conceptual procedure. Task variants and protocol versions retain their separate scoring conditions.","source_ids":["evidence-benchmark-proteinbench-snapshot"],"source_locator":"ProteinBench official website: Abstract; sequence/structure task evaluations; metric descriptions"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary paper and/or task implementation reviewed for the explicitly cited methodology claims. Scope-limited absence is recorded only after the documented source search; no model runs or independent reproduction."}}},"description":"Protein prediction, design and dynamics evaluation","facets":{"areas":["protein-structure"]},"id":"discovery-benchmark-proteinbench","kind":"benchmark","links":[],"name":"ProteinBench","source_ids":["src-discovery-proteinbench"],"status":"discovered"} {"attributes":{"entity_level":"suite","scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Protein variant effect prediction","version":null,"historical_missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"ProteinGym separates experimental variant-effect and clinical annotation tasks under supervised and zero-shot regimes.","summary_source_ids":["src-discovery-oatml-markslab-proteingym"],"summary_source_locator":"Pinned README: benchmark data; performance metrics and aggregation","sections":[{"title":"Evaluation methodology","body":"DMS assays and human clinical variants, with substitution and indel collections kept separate. Five-fold random, contiguous-position and modulo-position cross-validation are separate supervised DMS regimes. The original clinical analysis uses available ClinVar labels with explicit overlap warnings; zero-shot scoring does not fit on assay labels. Zero-shot DMS: Spearman, NDCG, AUC, MCC and top-K recall; supervised DMS: Spearman/MSE; clinical: AUC. Aggregation first groups assays by UniProt ID, then averages functional categories. Supervised DMS evaluations distinguish five-fold random, contiguous-position and modulo-position partitions. The original clinical benchmark explicitly warns that supervised methods may overlap ClinVar labels and that population-frequency training can leak information into benign-variant evaluation.","source_ids":["src-discovery-oatml-markslab-proteingym","evidence-discovery-final-proteingym"],"source_locator":"Pinned README: benchmark data; performance metrics and aggregation; Supervised DMS benchmarking; Sections on supervised DMS and clinical benchmarking"}],"facts":[{"label":"Datasets","value":"DMS assays and human clinical variants, with substitution and indel collections kept separate.","status":"source_checked","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"Pinned README: benchmark data; performance metrics and aggregation"},{"label":"Splits","value":"Five-fold random, contiguous-position and modulo-position cross-validation are separate supervised DMS regimes. The original clinical analysis uses available ClinVar labels with explicit overlap warnings; zero-shot scoring does not fit on assay labels.","status":"source_checked","source_ids":["evidence-discovery-final-proteingym"],"source_locator":"Supervised DMS benchmarking"},{"label":"Metrics","value":"Zero-shot DMS: Spearman, NDCG, AUC, MCC and top-K recall; supervised DMS: Spearman/MSE; clinical: AUC. Aggregation first groups assays by UniProt ID, then averages functional categories.","status":"source_checked","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"Pinned README: benchmark data; performance metrics and aggregation"},{"label":"Baselines","value":"The README distinguishes sequence-only baselines such as ESM-1v, alignment-based approaches such as DeepSequence/EVE, and sequence-plus-structure approaches such as SaProt; clinical baselines use dbNSFP 4.4a.","status":"source_checked","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"Pinned README: benchmark data; performance metrics and aggregation"},{"label":"Leakage controls","value":"Supervised DMS evaluations distinguish five-fold random, contiguous-position and modulo-position partitions. The original clinical benchmark explicitly warns that supervised methods may overlap ClinVar labels and that population-frequency training can leak information into benign-variant evaluation.","status":"source_checked","source_ids":["evidence-discovery-final-proteingym"],"source_locator":"Sections on supervised DMS and clinical benchmarking"},{"label":"Uncertainty","value":"Bootstrapped standard errors are provided for aggregate metrics.","status":"source_checked","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"Pinned README: benchmark data; performance metrics and aggregation"},{"label":"Entity type","value":"Benchmark suite with separate DMS/clinical, substitution/indel and supervised/zero-shot tracks.","status":"source_checked","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"Pinned README: benchmark data; performance metrics and aggregation"},{"label":"Organisms","value":"DMS collections span taxa; the clinical track concerns human proteins. Taxa-specific performance files are provided.","status":"source_checked","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"Pinned README: benchmark data; performance metrics and aggregation"},{"label":"Assays","value":"Deep mutational scanning measurements and curated benign/pathogenic clinical annotations.","status":"source_checked","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"Pinned README: benchmark data; performance metrics and aggregation"},{"label":"Allowed inputs","value":"Variant and target protein sequences; comparator modalities separately include alignments, structures and function annotations.","status":"source_checked","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"Pinned README: benchmark data; performance metrics and aggregation"},{"label":"Adaptation","value":"Separate zero-shot scoring and supervised learning regimes; labeled-data access must follow the chosen track.","status":"source_checked","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"Pinned README: benchmark data; performance metrics and aggregation"}],"strengths":[{"text":"Protein-level and functional-category aggregation reduces over-weighting of proteins with many assays.","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"Pinned README: benchmark data; performance metrics and aggregation"}],"limitations":[{"text":"ProteinGym’s DMS cross-validation controls and clinical-label overlap risks are different. Zero-shot, supervised, substitution and indel tasks must remain separate, with original assay identities preserved.","source_ids":["evidence-discovery-final-proteingym"],"source_locator":"Sections on supervised DMS and clinical benchmarking"}],"diagram":{"title":"Evaluation procedure","steps":["Allowed inputs: Variant and target protein sequences; comparator modalities separately include alignments, structures and function annotations.","Splits: Five-fold random, contiguous-position and modulo-position cross-validation are separate supervised DMS regimes. The original clinical analysis uses available ClinVar labels with explicit overlap warnings; zero-shot scoring does not fit on assay labels.","Metrics: Zero-shot DMS: Spearman, NDCG, AUC, MCC and top-K recall; supervised DMS: Spearman/MSE; clinical: AUC. Aggregation first groups assays by UniProt ID, then averages functional categories."],"caption":"Conceptual procedure. Task variants and protocol versions retain their separate scoring conditions.","source_ids":["src-discovery-oatml-markslab-proteingym","evidence-discovery-final-proteingym"],"source_locator":"Pinned README: benchmark data; performance metrics and aggregation; Supervised DMS benchmarking"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary paper and/or task implementation reviewed for the explicitly cited methodology claims. Scope-limited absence is recorded only after the documented source search; no model runs or independent reproduction."}}},"description":"Protein variant effect prediction","facets":{"areas":["protein-function"]},"id":"discovery-benchmark-proteingym","kind":"benchmark","links":[{"relation":"evaluates_task","target_id":"catalog-task-proteingym-effects"}],"name":"ProteinGym","source_ids":["src-discovery-oatml-markslab-proteingym"],"status":"discovered"} {"attributes":{"entity_level":"suite","scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Single-cell integration evaluation","version":null,"historical_missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"scIB evaluates single-cell integration by checking batch removal and biological conservation separately.","summary_source_ids":["src-discovery-theislab-scib"],"summary_source_locator":"Pinned README: package purpose; Metrics; Integration Tools","sections":[{"title":"Evaluation methodology","body":"Annotated single-cell data in AnnData form, with preprocessing and integration output types selected for the evaluation. The evaluator scores supplied outputs; predictor train/test partitions belong to the dataset/run being evaluated. The metric module separates batch-correction and biological-conservation measures. Listed integrations include Harmony, MNN/FastMNN, scVI/scANVI, Scanorama, BBKNN and Seurat. Uncertainty across samples, datasets or training runs must be defined by the evaluation study; this evaluator entry does not fix one experiment.","source_ids":["src-discovery-theislab-scib"],"source_locator":"Pinned README: package purpose; Metrics; Integration Tools"}],"facts":[{"label":"Datasets","value":"Annotated single-cell data in AnnData form, with preprocessing and integration output types selected for the evaluation.","status":"source_checked","source_ids":["src-discovery-theislab-scib"],"source_locator":"Pinned README: package purpose; Metrics; Integration Tools"},{"label":"Splits","value":"The evaluator scores supplied outputs; predictor train/test partitions belong to the dataset/run being evaluated.","status":"inapplicable","source_ids":["src-discovery-theislab-scib"],"source_locator":"Pinned README: package purpose; Metrics; Integration Tools"},{"label":"Metrics","value":"The metric module separates batch-correction and biological-conservation measures.","status":"source_checked","source_ids":["src-discovery-theislab-scib"],"source_locator":"Pinned README: package purpose; Metrics; Integration Tools"},{"label":"Baselines","value":"Listed integrations include Harmony, MNN/FastMNN, scVI/scANVI, Scanorama, BBKNN and Seurat.","status":"source_checked","source_ids":["src-discovery-theislab-scib"],"source_locator":"Pinned README: package purpose; Metrics; Integration Tools"},{"label":"Leakage controls","value":"scIB evaluates integration of the supplied batches together. Some methods use cell-type labels and others do not; the paper reports this distinction. This transductive integration setting is not a held-out-cell classifier test, so supervised train/test leakage terminology cannot be applied without specifying the method.","status":"source_checked","source_ids":["evidence-discovery-final-scib"],"source_locator":"Methods and Results: integration inputs, label use and biological-conservation metrics"},{"label":"Uncertainty","value":"Uncertainty across samples, datasets or training runs must be defined by the evaluation study; this evaluator entry does not fix one experiment.","status":"inapplicable","source_ids":["src-discovery-theislab-scib"],"source_locator":"Pinned README: package purpose; Metrics; Integration Tools"},{"label":"Entity type","value":"Single-cell integration evaluator.","status":"source_checked","source_ids":["src-discovery-theislab-scib"],"source_locator":"Pinned README: package purpose; Metrics; Integration Tools"},{"label":"Organisms","value":"Dataset choice supplies the organism; the metric package does not prescribe one.","status":"inapplicable","source_ids":["src-discovery-theislab-scib"],"source_locator":"Pinned README: package purpose; Metrics; Integration Tools"},{"label":"Assays","value":"Single-cell expression with batch and biological annotations.","status":"source_checked","source_ids":["src-discovery-theislab-scib"],"source_locator":"Pinned README: package purpose; Metrics; Integration Tools"},{"label":"Allowed inputs","value":"AnnData and integration outputs; label-dependent metrics additionally require biological labels.","status":"source_checked","source_ids":["src-discovery-theislab-scib"],"source_locator":"Pinned README: package purpose; Metrics; Integration Tools"},{"label":"Adaptation","value":"scIB scores integration outputs; adaptation occurs in the compared integration methods.","status":"inapplicable","source_ids":["src-discovery-theislab-scib"],"source_locator":"Pinned README: package purpose; Metrics; Integration Tools"}],"strengths":[{"text":"Separate batch-removal and biological-conservation scores reveal their trade-off.","source_ids":["src-discovery-theislab-scib"],"source_locator":"Pinned README: package purpose; Metrics; Integration Tools"}],"limitations":[{"text":"Batch mixing and preservation of biological variation must be considered together. Annotation-assisted methods and unsupervised methods receive different input information, and integration of known batches does not establish transfer to unseen batches.","source_ids":["evidence-discovery-final-scib"],"source_locator":"Methods and Results: integration inputs, label use and biological-conservation metrics"}],"diagram":{"title":"Evaluation procedure","steps":["Allowed inputs: AnnData and integration outputs; label-dependent metrics additionally require biological labels.","Splits: The evaluator scores supplied outputs; predictor train/test partitions belong to the dataset/run being evaluated.","Metrics: The metric module separates batch-correction and biological-conservation measures."],"caption":"Conceptual procedure. Task variants and protocol versions retain their separate scoring conditions.","source_ids":["src-discovery-theislab-scib"],"source_locator":"Pinned README: package purpose; Metrics; Integration Tools"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary paper and/or task implementation reviewed for the explicitly cited methodology claims. Scope-limited absence is recorded only after the documented source search; no model runs or independent reproduction."}}},"description":"Single-cell integration evaluation","facets":{"areas":["single-cell"]},"id":"discovery-benchmark-scib","kind":"benchmark","links":[{"relation":"evaluates_task","target_id":"catalog-task-cell-batch-integration"}],"name":"scIB","source_ids":["src-discovery-theislab-scib"],"status":"discovered"} {"attributes":{"entity_level":"evaluator","scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Perturbation prediction metric calibration","version":null,"historical_missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"scPertEval evaluates and calibrates scoring protocols for single-cell perturbation predictions.","summary_source_ids":["src-discovery-virtual-cell-research-community-scperteval"],"summary_source_locator":"Pinned README: toolkit purpose; scoring/calibration/DE actions; protocol taxonomy","sections":[{"title":"Evaluation methodology","body":"scPertEval supplies explicit scoring protocols for perturbation predictions. A metric can compare one perturbation or operate across the complete perturbation panel, depending on its declared scope. Ground-truth references and context are passed to the evaluator, so protocol and preprocessing choices remain part of the result.","source_ids":["evidence-discovery-final-scperteval0"],"source_locator":"Pinned src/scperteval/protocols/metrics.py: metric input contract and context"}],"facts":[{"label":"Datasets","value":"Predicted and observed perturbation responses; the associated study evaluates protocols across public datasets.","status":"source_checked","source_ids":["src-discovery-virtual-cell-research-community-scperteval"],"source_locator":"Pinned README: toolkit purpose; scoring/calibration/DE actions; protocol taxonomy"},{"label":"Splits","value":"The evaluator scores supplied outputs; predictor train/test partitions belong to the dataset/run being evaluated.","status":"inapplicable","source_ids":["src-discovery-virtual-cell-research-community-scperteval"],"source_locator":"Pinned README: toolkit purpose; scoring/calibration/DE actions; protocol taxonomy"},{"label":"Metrics","value":"Protocol scoring is separated from calibration against empirical positive/negative controls using DRF and BDS.","status":"source_checked","source_ids":["src-discovery-virtual-cell-research-community-scperteval"],"source_locator":"Pinned README: toolkit purpose; scoring/calibration/DE actions; protocol taxonomy"},{"label":"Baselines","value":"Empirical controls calibrate how well a protocol distinguishes expected response quality.","status":"source_checked","source_ids":["src-discovery-virtual-cell-research-community-scperteval"],"source_locator":"Pinned README: toolkit purpose; scoring/calibration/DE actions; protocol taxonomy"},{"label":"Leakage controls","value":"The metric implementation receives ground truth, predictions and an evaluation context; it does not construct model-training partitions or audit training data. Predictor leakage controls belong to the protocol and dataset used to produce the submitted predictions.","status":"inapplicable","source_ids":["evidence-discovery-final-scperteval0"],"source_locator":"Pinned src/scperteval/protocols/metrics.py: metric input contract and context"},{"label":"Uncertainty","value":"Uncertainty across samples, datasets or training runs must be defined by the evaluation study; this evaluator entry does not fix one experiment.","status":"inapplicable","source_ids":["src-discovery-virtual-cell-research-community-scperteval"],"source_locator":"Pinned README: toolkit purpose; scoring/calibration/DE actions; protocol taxonomy"},{"label":"Entity type","value":"Perturbation scoring-protocol calibration toolkit.","status":"source_checked","source_ids":["src-discovery-virtual-cell-research-community-scperteval"],"source_locator":"Pinned README: toolkit purpose; scoring/calibration/DE actions; protocol taxonomy"},{"label":"Organisms","value":"Organism scope belongs to the selected perturbation dataset.","status":"inapplicable","source_ids":["src-discovery-virtual-cell-research-community-scperteval"],"source_locator":"Pinned README: toolkit purpose; scoring/calibration/DE actions; protocol taxonomy"},{"label":"Assays","value":"Observed single-cell perturbation responses and empirical positive/negative controls.","status":"source_checked","source_ids":["src-discovery-virtual-cell-research-community-scperteval"],"source_locator":"Pinned README: toolkit purpose; scoring/calibration/DE actions; protocol taxonomy"},{"label":"Allowed inputs","value":"Predicted/observed responses plus a protocol specifying representation, metric, transformation and reporting.","status":"source_checked","source_ids":["src-discovery-virtual-cell-research-community-scperteval"],"source_locator":"Pinned README: toolkit purpose; scoring/calibration/DE actions; protocol taxonomy"},{"label":"Adaptation","value":"The toolkit scores and calibrates evaluation protocols; it does not impose predictor fine-tuning.","status":"inapplicable","source_ids":["src-discovery-virtual-cell-research-community-scperteval"],"source_locator":"Pinned README: toolkit purpose; scoring/calibration/DE actions; protocol taxonomy"}],"strengths":[{"text":"Calibration tests whether a score distinguishes empirical controls before using it to rank models.","source_ids":["src-discovery-virtual-cell-research-community-scperteval"],"source_locator":"Pinned README: toolkit purpose; scoring/calibration/DE actions; protocol taxonomy"}],"limitations":[{"text":"Metric computation alone cannot establish whether a predictor saw a held-out perturbation or cell context during training. That evidence must come from the model and dataset protocol.","source_ids":["evidence-discovery-final-scperteval0"],"source_locator":"Pinned src/scperteval/protocols/metrics.py: metric input contract and context"}],"diagram":{"title":"Evaluation procedure","steps":["Allowed inputs: Predicted/observed responses plus a protocol specifying representation, metric, transformation and reporting.","Splits: The evaluator scores supplied outputs; predictor train/test partitions belong to the dataset/run being evaluated.","Metrics: Protocol scoring is separated from calibration against empirical positive/negative controls using DRF and BDS."],"caption":"Conceptual procedure. Task variants and protocol versions retain their separate scoring conditions.","source_ids":["src-discovery-virtual-cell-research-community-scperteval"],"source_locator":"Pinned README: toolkit purpose; scoring/calibration/DE actions; protocol taxonomy"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary paper and/or task implementation reviewed for the explicitly cited methodology claims. Scope-limited absence is recorded only after the documented source search; no model runs or independent reproduction."}}},"description":"Perturbation prediction metric calibration","facets":{"areas":["single-cell"]},"id":"discovery-benchmark-scperteval","kind":"benchmark","links":[],"name":"scPertEval","source_ids":["src-discovery-virtual-cell-research-community-scperteval"],"status":"discovered"} {"attributes":{"entity_level":"suite","scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Protein representation learning tasks","version":null,"historical_missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"TAPE evaluates protein representations through five supervised downstream tasks.","summary_source_ids":["src-discovery-songlab-cal-tape"],"summary_source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards","sections":[{"title":"Evaluation methodology","body":"TAPE evaluates protein representations on five tasks: local secondary structure, contacts, remote homology, fluorescence and stability. Each has a biologically motivated supervised partition, and the original paper compares pretrained representations with untrained and alignment-based controls using task-specific prediction heads.","source_ids":["evidence-discovery-final-tape"],"source_locator":"Section 3; Appendix A.1"}],"facts":[{"label":"Datasets","value":"Secondary structure, contacts, remote homology, fluorescence and stability; a Pfam pretraining corpus is supplied separately.","status":"source_checked","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards"},{"label":"Splits","value":"Secondary structure uses 25% identity filtering; contacts use ProteinNet/CASP12 with 30% filtering; remote homology holds out superfamilies; fluorescence holds out greater mutation distances; stability holds out selected mutation neighbourhoods.","status":"source_checked","source_ids":["evidence-discovery-final-tape"],"source_locator":"Section 3; Appendix A.1"},{"label":"Metrics","value":"Task leaderboards use three-class accuracy, contact ranking, top-1 homology accuracy and Spearman correlation for fluorescence/stability.","status":"source_checked","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards"},{"label":"Baselines","value":"Transformer, LSTM, ResNet, UniRep and one-hot baselines.","status":"source_checked","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards"},{"label":"Leakage controls","value":"The five tasks use different supervised generalization boundaries. Protein identity, evolutionary groups and mutational distance are not interchangeable, and none is a universal audit of unsupervised pretraining overlap.","status":"source_checked","source_ids":["evidence-discovery-final-tape"],"source_locator":"Section 3; Appendix A.1"},{"label":"Uncertainty","value":"The original paper reports point estimates in its task result tables. Methods and task appendices do not define a suite-wide repeated-seed or bootstrap interval; individual later evaluations must supply their own uncertainty.","status":"unreported","source_ids":["evidence-discovery-final-tape"],"source_locator":"Section 3; Appendix A.1; Tables 1–2"},{"label":"Entity type","value":"Protein-representation benchmark suite.","status":"source_checked","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards"},{"label":"Organisms","value":"Mixed protein-domain and structural sources rather than a species-held-out benchmark; the engineering tasks use green fluorescent protein variants and designed stability landscapes.","status":"source_checked","source_ids":["evidence-discovery-final-tape"],"source_locator":"Section 3; Appendix A.1; Tables 1–2"},{"label":"Assays","value":"Protein structure/homology annotations and experimental fluorescence/stability measurements.","status":"source_checked","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards"},{"label":"Allowed inputs","value":"Protein amino-acid sequences and task-specific labels.","status":"source_checked","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards"},{"label":"Adaptation","value":"Unsupervised pretraining followed by supervised downstream training; the README warns that downstream hyperparameters require task-specific tuning.","status":"source_checked","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards"}],"strengths":[{"text":"Distinct sequence, residue and protein-level tasks avoid relying on language-model perplexity as a proxy for transfer.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards"}],"limitations":[{"text":"The original TAPE paper and the later PyTorch implementation are distinct versions. Preserve the task split and implementation; pretraining exposure is separate from supervised sequence-identity filtering.","source_ids":["evidence-discovery-final-tape"],"source_locator":"Section 3; Appendix A.1; Tables 1–2"}],"diagram":{"title":"Evaluation procedure","steps":["Allowed inputs: Protein amino-acid sequences and task-specific labels.","Splits: Secondary structure uses 25% identity filtering; contacts use ProteinNet/CASP12 with 30% filtering; remote homology holds out superfamilies; fluorescence holds out greater mutation distances; stability holds out selected mutation neighbourhoods.","Metrics: Task leaderboards use three-class accuracy, contact ranking, top-1 homology accuracy and Spearman correlation for fluorescence/stability."],"caption":"Conceptual procedure. Task variants and protocol versions retain their separate scoring conditions.","source_ids":["src-discovery-songlab-cal-tape","evidence-discovery-final-tape"],"source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards; Section 3; Appendix A.1"},"coverage":"limited","gaps":["Uncertainty: The original paper reports point estimates in its task result tables. Methods and task appendices do not define a suite-wide repeated-seed or bootstrap interval; individual later evaluations must supply their own uncertainty."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary paper and/or task implementation reviewed for the explicitly cited methodology claims. Scope-limited absence is recorded only after the documented source search; no model runs or independent reproduction."}}},"description":"Protein representation learning tasks","facets":{"areas":["protein-function"]},"id":"discovery-benchmark-tape","kind":"benchmark","links":[],"name":"TAPE","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"entity_level":"task","scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Contact Prediction","version":null,"historical_missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"The TAPE protein residue-contact prediction task evaluates a trained protein representation.","summary_source_ids":["src-discovery-songlab-cal-tape"],"summary_source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards; README task-specific leaderboard and data-source references","sections":[{"title":"Evaluation methodology","body":"ProteinNet training/validation partitions filtered at 30% sequence identity; the held-out evaluation uses CASP12 targets. Sequence-identity filtering and the CASP12 target set define the supervised generalization boundary.","source_ids":["evidence-discovery-final-tape"],"source_locator":"Section 3 task definitions; Appendix A.1.1–A.1.5"}],"facts":[{"label":"Datasets","value":"ProteinNet structural data with CASP12 test targets.","status":"source_checked","source_ids":["evidence-discovery-final-tape"],"source_locator":"Appendix A.1"},{"label":"Splits","value":"ProteinNet training/validation partitions filtered at 30% sequence identity; the held-out evaluation uses CASP12 targets.","status":"source_checked","source_ids":["evidence-discovery-final-tape"],"source_locator":"Section 3 task definitions; Appendix A.1.1–A.1.5"},{"label":"Metrics","value":"Precision at L/5 for medium- and long-range contacts.","status":"source_checked","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards; README task-specific leaderboard and data-source references"},{"label":"Baselines","value":"Task leaderboard comparisons include Transformer, LSTM, UniRep, ResNet, Bepler and one-hot baselines; alignment-augmented references appear where applicable.","status":"source_checked","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards; README task-specific leaderboard and data-source references"},{"label":"Leakage controls","value":"Sequence-identity filtering and the CASP12 target set define the supervised generalization boundary.","status":"source_checked","source_ids":["evidence-discovery-final-tape"],"source_locator":"Section 3 task definitions; Appendix A.1.1–A.1.5"},{"label":"Uncertainty","value":"The original paper reports point estimates in its task result tables. Methods and task appendices do not define a suite-wide repeated-seed or bootstrap interval; individual later evaluations must supply their own uncertainty.","status":"unreported","source_ids":["evidence-discovery-final-tape"],"source_locator":"Section 3; Appendix A.1; Tables 1–2"},{"label":"Entity type","value":"Constituent benchmark task: TAPE Contact Prediction","status":"source_checked","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards; README task-specific leaderboard and data-source references"},{"label":"Organisms","value":"Mixed-organism ProteinNet/CASP protein structures; organism identity is not the partitioning key.","status":"source_checked","source_ids":["evidence-discovery-final-tape"],"source_locator":"Section 3 task definitions; Appendix A.1.1–A.1.5"},{"label":"Assays","value":"Protein structure/homology annotations and experimental fluorescence/stability measurements.","status":"source_checked","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards; README task-specific leaderboard and data-source references"},{"label":"Allowed inputs","value":"Protein sequence with a residue-contact target derived from ProteinNet.","status":"source_checked","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards; README task-specific leaderboard and data-source references"},{"label":"Adaptation","value":"Unsupervised pretraining followed by supervised downstream training; the README warns that downstream hyperparameters require task-specific tuning.","status":"source_checked","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards; README task-specific leaderboard and data-source references"}],"strengths":[{"text":"Distinct sequence, residue and protein-level tasks avoid relying on language-model perplexity as a proxy for transfer.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards; README task-specific leaderboard and data-source references"}],"limitations":[{"text":"The original TAPE paper and the later PyTorch implementation are distinct versions. Preserve the task split and implementation; pretraining exposure is separate from supervised sequence-identity filtering.","source_ids":["evidence-discovery-final-tape"],"source_locator":"Section 3; Appendix A.1; Tables 1–2"}],"diagram":{"title":"Evaluation procedure","steps":["Allowed inputs: Protein sequence with a residue-contact target derived from ProteinNet.","Splits: ProteinNet training/validation partitions filtered at 30% sequence identity; the held-out evaluation uses CASP12 targets.","Metrics: Precision at L/5 for medium- and long-range contacts."],"caption":"Conceptual procedure. Task variants and protocol versions retain their separate scoring conditions.","source_ids":["src-discovery-songlab-cal-tape","evidence-discovery-final-tape"],"source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards; README task-specific leaderboard and data-source references; Section 3 task definitions; Appendix A.1.1–A.1.5"},"coverage":"limited","gaps":["Uncertainty: The original paper reports point estimates in its task result tables. Methods and task appendices do not define a suite-wide repeated-seed or bootstrap interval; individual later evaluations must supply their own uncertainty."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary paper and/or task implementation reviewed for the explicitly cited methodology claims. Scope-limited absence is recorded only after the documented source search; no model runs or independent reproduction."}}},"description":"Contact Prediction","facets":{"areas":["protein-function"]},"id":"discovery-benchmark-tape-contact-prediction","kind":"benchmark","links":[{"relation":"parent","target_id":"discovery-benchmark-tape"},{"relation":"part_of","target_id":"discovery-benchmark-tape"}],"name":"TAPE Contact Prediction","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"entity_level":"task","scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Fluorescence","version":null,"historical_missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"The TAPE protein fluorescence prediction task evaluates a trained protein representation.","summary_source_ids":["src-discovery-songlab-cal-tape"],"summary_source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards; README task-specific leaderboard and data-source references","sections":[{"title":"Evaluation methodology","body":"Training and validation use GFP variants within three mutations of the parent; testing uses variants with four to fifteen mutations. Mutation-distance separation tests extrapolation away from the same parent protein rather than independence of protein families.","source_ids":["evidence-discovery-final-tape"],"source_locator":"Section 3 task definitions; Appendix A.1.1–A.1.5"}],"facts":[{"label":"Datasets","value":"Sarkisyan green fluorescent protein mutagenesis assay, using the original TAPE mutation-distance partition.","status":"source_checked","source_ids":["evidence-discovery-final-tape"],"source_locator":"Appendix A.1"},{"label":"Splits","value":"Training and validation use GFP variants within three mutations of the parent; testing uses variants with four to fifteen mutations.","status":"source_checked","source_ids":["evidence-discovery-final-tape"],"source_locator":"Section 3 task definitions; Appendix A.1.1–A.1.5"},{"label":"Metrics","value":"Spearman rank correlation.","status":"source_checked","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards; README task-specific leaderboard and data-source references"},{"label":"Baselines","value":"Task leaderboard comparisons include Transformer, LSTM, UniRep, ResNet, Bepler and one-hot baselines; alignment-augmented references appear where applicable.","status":"source_checked","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards; README task-specific leaderboard and data-source references"},{"label":"Leakage controls","value":"Mutation-distance separation tests extrapolation away from the same parent protein rather than independence of protein families.","status":"source_checked","source_ids":["evidence-discovery-final-tape"],"source_locator":"Section 3 task definitions; Appendix A.1.1–A.1.5"},{"label":"Uncertainty","value":"The original paper reports point estimates in its task result tables. Methods and task appendices do not define a suite-wide repeated-seed or bootstrap interval; individual later evaluations must supply their own uncertainty.","status":"unreported","source_ids":["evidence-discovery-final-tape"],"source_locator":"Section 3; Appendix A.1; Tables 1–2"},{"label":"Entity type","value":"Constituent benchmark task: TAPE Fluorescence","status":"source_checked","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards; README task-specific leaderboard and data-source references"},{"label":"Organisms","value":"Variants of the parent green fluorescent protein in the Sarkisyan mutagenesis assay.","status":"source_checked","source_ids":["evidence-discovery-final-tape"],"source_locator":"Section 3 task definitions; Appendix A.1.1–A.1.5"},{"label":"Assays","value":"Protein structure/homology annotations and experimental fluorescence/stability measurements.","status":"source_checked","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards; README task-specific leaderboard and data-source references"},{"label":"Allowed inputs","value":"Protein sequence with a measured fluorescence target.","status":"source_checked","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards; README task-specific leaderboard and data-source references"},{"label":"Adaptation","value":"Unsupervised pretraining followed by supervised downstream training; the README warns that downstream hyperparameters require task-specific tuning.","status":"source_checked","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards; README task-specific leaderboard and data-source references"}],"strengths":[{"text":"Distinct sequence, residue and protein-level tasks avoid relying on language-model perplexity as a proxy for transfer.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards; README task-specific leaderboard and data-source references"}],"limitations":[{"text":"The original TAPE paper and the later PyTorch implementation are distinct versions. Preserve the task split and implementation; pretraining exposure is separate from supervised sequence-identity filtering.","source_ids":["evidence-discovery-final-tape"],"source_locator":"Section 3; Appendix A.1; Tables 1–2"}],"diagram":{"title":"Evaluation procedure","steps":["Allowed inputs: Protein sequence with a measured fluorescence target.","Splits: Training and validation use GFP variants within three mutations of the parent; testing uses variants with four to fifteen mutations.","Metrics: Spearman rank correlation."],"caption":"Conceptual procedure. Task variants and protocol versions retain their separate scoring conditions.","source_ids":["src-discovery-songlab-cal-tape","evidence-discovery-final-tape"],"source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards; README task-specific leaderboard and data-source references; Section 3 task definitions; Appendix A.1.1–A.1.5"},"coverage":"limited","gaps":["Uncertainty: The original paper reports point estimates in its task result tables. Methods and task appendices do not define a suite-wide repeated-seed or bootstrap interval; individual later evaluations must supply their own uncertainty."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary paper and/or task implementation reviewed for the explicitly cited methodology claims. Scope-limited absence is recorded only after the documented source search; no model runs or independent reproduction."}}},"description":"Fluorescence","facets":{"areas":["protein-function"]},"id":"discovery-benchmark-tape-fluorescence","kind":"benchmark","links":[{"relation":"parent","target_id":"discovery-benchmark-tape"},{"relation":"part_of","target_id":"discovery-benchmark-tape"}],"name":"TAPE Fluorescence","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"entity_level":"task","scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Remote Homology Detection","version":null,"historical_missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"The TAPE remote protein-homology classification task evaluates a trained protein representation.","summary_source_ids":["src-discovery-songlab-cal-tape"],"summary_source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards; README task-specific leaderboard and data-source references","sections":[{"title":"Evaluation methodology","body":"SCOP 1.75 protein domains are grouped by evolutionary hierarchy; entire superfamilies are held out for fold-level classification. Holding out superfamilies tests remote homologues without transferring examples from the same superfamily between train and test.","source_ids":["evidence-discovery-final-tape"],"source_locator":"Section 3 task definitions; Appendix A.1.1–A.1.5"}],"facts":[{"label":"Datasets","value":"SCOP 1.75 domains with fold labels and held-out superfamilies.","status":"source_checked","source_ids":["evidence-discovery-final-tape"],"source_locator":"Appendix A.1"},{"label":"Splits","value":"SCOP 1.75 protein domains are grouped by evolutionary hierarchy; entire superfamilies are held out for fold-level classification.","status":"source_checked","source_ids":["evidence-discovery-final-tape"],"source_locator":"Section 3 task definitions; Appendix A.1.1–A.1.5"},{"label":"Metrics","value":"Top-1 class accuracy.","status":"source_checked","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards; README task-specific leaderboard and data-source references"},{"label":"Baselines","value":"Task leaderboard comparisons include Transformer, LSTM, UniRep, ResNet, Bepler and one-hot baselines; alignment-augmented references appear where applicable.","status":"source_checked","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards; README task-specific leaderboard and data-source references"},{"label":"Leakage controls","value":"Holding out superfamilies tests remote homologues without transferring examples from the same superfamily between train and test.","status":"source_checked","source_ids":["evidence-discovery-final-tape"],"source_locator":"Section 3 task definitions; Appendix A.1.1–A.1.5"},{"label":"Uncertainty","value":"The original paper reports point estimates in its task result tables. Methods and task appendices do not define a suite-wide repeated-seed or bootstrap interval; individual later evaluations must supply their own uncertainty.","status":"unreported","source_ids":["evidence-discovery-final-tape"],"source_locator":"Section 3; Appendix A.1; Tables 1–2"},{"label":"Entity type","value":"Constituent benchmark task: TAPE Remote Homology Detection","status":"source_checked","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards; README task-specific leaderboard and data-source references"},{"label":"Organisms","value":"SCOP protein-domain collection across organisms; the evaluated label is structural fold, not species.","status":"source_checked","source_ids":["evidence-discovery-final-tape"],"source_locator":"Section 3 task definitions; Appendix A.1.1–A.1.5"},{"label":"Assays","value":"Protein structure/homology annotations and experimental fluorescence/stability measurements.","status":"source_checked","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards; README task-specific leaderboard and data-source references"},{"label":"Allowed inputs","value":"Protein sequence with a remote-homology class.","status":"source_checked","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards; README task-specific leaderboard and data-source references"},{"label":"Adaptation","value":"Unsupervised pretraining followed by supervised downstream training; the README warns that downstream hyperparameters require task-specific tuning.","status":"source_checked","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards; README task-specific leaderboard and data-source references"}],"strengths":[{"text":"Distinct sequence, residue and protein-level tasks avoid relying on language-model perplexity as a proxy for transfer.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards; README task-specific leaderboard and data-source references"}],"limitations":[{"text":"The original TAPE paper and the later PyTorch implementation are distinct versions. Preserve the task split and implementation; pretraining exposure is separate from supervised sequence-identity filtering.","source_ids":["evidence-discovery-final-tape"],"source_locator":"Section 3; Appendix A.1; Tables 1–2"}],"diagram":{"title":"Evaluation procedure","steps":["Allowed inputs: Protein sequence with a remote-homology class.","Splits: SCOP 1.75 protein domains are grouped by evolutionary hierarchy; entire superfamilies are held out for fold-level classification.","Metrics: Top-1 class accuracy."],"caption":"Conceptual procedure. Task variants and protocol versions retain their separate scoring conditions.","source_ids":["src-discovery-songlab-cal-tape","evidence-discovery-final-tape"],"source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards; README task-specific leaderboard and data-source references; Section 3 task definitions; Appendix A.1.1–A.1.5"},"coverage":"limited","gaps":["Uncertainty: The original paper reports point estimates in its task result tables. Methods and task appendices do not define a suite-wide repeated-seed or bootstrap interval; individual later evaluations must supply their own uncertainty."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary paper and/or task implementation reviewed for the explicitly cited methodology claims. Scope-limited absence is recorded only after the documented source search; no model runs or independent reproduction."}}},"description":"Remote Homology Detection","facets":{"areas":["protein-function"]},"id":"discovery-benchmark-tape-remote-homology-detection","kind":"benchmark","links":[{"relation":"parent","target_id":"discovery-benchmark-tape"},{"relation":"part_of","target_id":"discovery-benchmark-tape"}],"name":"TAPE Remote Homology Detection","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"entity_level":"task","scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Secondary Structure","version":null,"historical_missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"The TAPE residue secondary-structure classification task evaluates a trained protein representation.","summary_source_ids":["src-discovery-songlab-cal-tape"],"summary_source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards; README task-specific leaderboard and data-source references","sections":[{"title":"Evaluation methodology","body":"Training/validation and CB513, CASP12 and TS115 test proteins are filtered at 25% sequence identity. Identity filtering excludes close train/test homologues; it is not a species holdout.","source_ids":["evidence-discovery-final-tape"],"source_locator":"Section 3 task definitions; Appendix A.1.1–A.1.5"}],"facts":[{"label":"Datasets","value":"Klausen training/validation proteins with CB513, CASP12 and TS115 structure-label test collections.","status":"source_checked","source_ids":["evidence-discovery-final-tape"],"source_locator":"Appendix A.1"},{"label":"Splits","value":"Training/validation and CB513, CASP12 and TS115 test proteins are filtered at 25% sequence identity.","status":"source_checked","source_ids":["evidence-discovery-final-tape"],"source_locator":"Section 3 task definitions; Appendix A.1.1–A.1.5"},{"label":"Metrics","value":"Three-class residue accuracy.","status":"source_checked","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards; README task-specific leaderboard and data-source references"},{"label":"Baselines","value":"Task leaderboard comparisons include Transformer, LSTM, UniRep, ResNet, Bepler and one-hot baselines; alignment-augmented references appear where applicable.","status":"source_checked","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards; README task-specific leaderboard and data-source references"},{"label":"Leakage controls","value":"Identity filtering excludes close train/test homologues; it is not a species holdout.","status":"source_checked","source_ids":["evidence-discovery-final-tape"],"source_locator":"Section 3 task definitions; Appendix A.1.1–A.1.5"},{"label":"Uncertainty","value":"The original paper reports point estimates in its task result tables. Methods and task appendices do not define a suite-wide repeated-seed or bootstrap interval; individual later evaluations must supply their own uncertainty.","status":"unreported","source_ids":["evidence-discovery-final-tape"],"source_locator":"Section 3; Appendix A.1; Tables 1–2"},{"label":"Entity type","value":"Constituent benchmark task: TAPE Secondary Structure","status":"source_checked","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards; README task-specific leaderboard and data-source references"},{"label":"Organisms","value":"Protein structures from the mixed-organism Klausen/CB513/CASP12/TS115 collections.","status":"source_checked","source_ids":["evidence-discovery-final-tape"],"source_locator":"Section 3 task definitions; Appendix A.1.1–A.1.5"},{"label":"Assays","value":"Protein structure/homology annotations and experimental fluorescence/stability measurements.","status":"source_checked","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards; README task-specific leaderboard and data-source references"},{"label":"Allowed inputs","value":"Protein amino-acid sequence with residue-level secondary-structure labels.","status":"source_checked","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards; README task-specific leaderboard and data-source references"},{"label":"Adaptation","value":"Unsupervised pretraining followed by supervised downstream training; the README warns that downstream hyperparameters require task-specific tuning.","status":"source_checked","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards; README task-specific leaderboard and data-source references"}],"strengths":[{"text":"Distinct sequence, residue and protein-level tasks avoid relying on language-model perplexity as a proxy for transfer.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards; README task-specific leaderboard and data-source references"}],"limitations":[{"text":"The original TAPE paper and the later PyTorch implementation are distinct versions. Preserve the task split and implementation; pretraining exposure is separate from supervised sequence-identity filtering.","source_ids":["evidence-discovery-final-tape"],"source_locator":"Section 3; Appendix A.1; Tables 1–2"}],"diagram":{"title":"Evaluation procedure","steps":["Allowed inputs: Protein amino-acid sequence with residue-level secondary-structure labels.","Splits: Training/validation and CB513, CASP12 and TS115 test proteins are filtered at 25% sequence identity.","Metrics: Three-class residue accuracy."],"caption":"Conceptual procedure. Task variants and protocol versions retain their separate scoring conditions.","source_ids":["src-discovery-songlab-cal-tape","evidence-discovery-final-tape"],"source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards; README task-specific leaderboard and data-source references; Section 3 task definitions; Appendix A.1.1–A.1.5"},"coverage":"limited","gaps":["Uncertainty: The original paper reports point estimates in its task result tables. Methods and task appendices do not define a suite-wide repeated-seed or bootstrap interval; individual later evaluations must supply their own uncertainty."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary paper and/or task implementation reviewed for the explicitly cited methodology claims. Scope-limited absence is recorded only after the documented source search; no model runs or independent reproduction."}}},"description":"Secondary Structure","facets":{"areas":["protein-function"]},"id":"discovery-benchmark-tape-secondary-structure","kind":"benchmark","links":[{"relation":"parent","target_id":"discovery-benchmark-tape"},{"relation":"part_of","target_id":"discovery-benchmark-tape"}],"name":"TAPE Secondary Structure","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"entity_level":"task","scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Stability","version":null,"historical_missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"The TAPE protein stability prediction task evaluates a trained protein representation.","summary_source_ids":["src-discovery-songlab-cal-tape"],"summary_source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards; README task-specific leaderboard and data-source references","sections":[{"title":"Evaluation methodology","body":"Training/validation use four rounds of designed-protein stability experiments; testing uses seventeen one-mutation neighbourhoods around selected promising proteins. Designed round-to-neighbourhood generalization is intentional; test variants can be close to parent proteins observed during training.","source_ids":["evidence-discovery-final-tape"],"source_locator":"Section 3 task definitions; Appendix A.1.1–A.1.5"}],"facts":[{"label":"Datasets","value":"Rocklin designed-protein stability measurements, using the original TAPE round-to-neighbourhood partition.","status":"source_checked","source_ids":["evidence-discovery-final-tape"],"source_locator":"Appendix A.1"},{"label":"Splits","value":"Training/validation use four rounds of designed-protein stability experiments; testing uses seventeen one-mutation neighbourhoods around selected promising proteins.","status":"source_checked","source_ids":["evidence-discovery-final-tape"],"source_locator":"Section 3 task definitions; Appendix A.1.1–A.1.5"},{"label":"Metrics","value":"Spearman rank correlation.","status":"source_checked","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards; README task-specific leaderboard and data-source references"},{"label":"Baselines","value":"Task leaderboard comparisons include Transformer, LSTM, UniRep, ResNet, Bepler and one-hot baselines; alignment-augmented references appear where applicable.","status":"source_checked","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards; README task-specific leaderboard and data-source references"},{"label":"Leakage controls","value":"Designed round-to-neighbourhood generalization is intentional; test variants can be close to parent proteins observed during training.","status":"source_checked","source_ids":["evidence-discovery-final-tape"],"source_locator":"Section 3 task definitions; Appendix A.1.1–A.1.5"},{"label":"Uncertainty","value":"The original paper reports point estimates in its task result tables. Methods and task appendices do not define a suite-wide repeated-seed or bootstrap interval; individual later evaluations must supply their own uncertainty.","status":"unreported","source_ids":["evidence-discovery-final-tape"],"source_locator":"Section 3; Appendix A.1; Tables 1–2"},{"label":"Entity type","value":"Constituent benchmark task: TAPE Stability","status":"source_checked","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards; README task-specific leaderboard and data-source references"},{"label":"Organisms","value":"Designed protein sequences measured by the Rocklin stability assay; a single natural organism label is inapplicable.","status":"source_checked","source_ids":["evidence-discovery-final-tape"],"source_locator":"Section 3 task definitions; Appendix A.1.1–A.1.5"},{"label":"Assays","value":"Protein structure/homology annotations and experimental fluorescence/stability measurements.","status":"source_checked","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards; README task-specific leaderboard and data-source references"},{"label":"Allowed inputs","value":"Protein sequence with a measured stability target.","status":"source_checked","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards; README task-specific leaderboard and data-source references"},{"label":"Adaptation","value":"Unsupervised pretraining followed by supervised downstream training; the README warns that downstream hyperparameters require task-specific tuning.","status":"source_checked","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards; README task-specific leaderboard and data-source references"}],"strengths":[{"text":"Distinct sequence, residue and protein-level tasks avoid relying on language-model perplexity as a proxy for transfer.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards; README task-specific leaderboard and data-source references"}],"limitations":[{"text":"The original TAPE paper and the later PyTorch implementation are distinct versions. Preserve the task split and implementation; pretraining exposure is separate from supervised sequence-identity filtering.","source_ids":["evidence-discovery-final-tape"],"source_locator":"Section 3; Appendix A.1; Tables 1–2"}],"diagram":{"title":"Evaluation procedure","steps":["Allowed inputs: Protein sequence with a measured stability target.","Splits: Training/validation use four rounds of designed-protein stability experiments; testing uses seventeen one-mutation neighbourhoods around selected promising proteins.","Metrics: Spearman rank correlation."],"caption":"Conceptual procedure. Task variants and protocol versions retain their separate scoring conditions.","source_ids":["src-discovery-songlab-cal-tape","evidence-discovery-final-tape"],"source_locator":"Pinned README: compatibility notice; Evaluating a Downstream Model; Data; task leaderboards; README task-specific leaderboard and data-source references; Section 3 task definitions; Appendix A.1.1–A.1.5"},"coverage":"limited","gaps":["Uncertainty: The original paper reports point estimates in its task result tables. Methods and task appendices do not define a suite-wide repeated-seed or bootstrap interval; individual later evaluations must supply their own uncertainty."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary paper and/or task implementation reviewed for the explicitly cited methodology claims. Scope-limited absence is recorded only after the documented source search; no model runs or independent reproduction."}}},"description":"Stability","facets":{"areas":["protein-function"]},"id":"discovery-benchmark-tape-stability","kind":"benchmark","links":[{"relation":"parent","target_id":"discovery-benchmark-tape"},{"relation":"part_of","target_id":"discovery-benchmark-tape"}],"name":"TAPE Stability","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"entity_level":"suite","scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Molecular binding, biochemical activity and related specialist tasks","version":null,"historical_missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"TDC organizes molecular prediction tasks into datasets and benchmark groups with explicit splitting and evaluation interfaces.","summary_source_ids":["src-discovery-mims-harvard-tdc"],"summary_source_locator":"Pinned README: data functions; dataset splits; evaluators; benchmark groups","sections":[{"title":"Evaluation methodology","body":"Therapeutics Data Commons supplies separate molecular tasks and curated benchmark groups. Each group specifies datasets, prediction units, partitions and metrics; the original ADMET example uses scaffold splits and simple descriptor or sequence baselines. Only the molecular and mechanistic tasks within rewire’s scope belong in this catalogue.","source_ids":["evidence-discovery-final-tdc"],"source_locator":"Section 9 and Tables 3–4: benchmark groups"}],"facts":[{"label":"Datasets","value":"Multiple task-specific molecular datasets, including grouped benchmark themes.","status":"source_checked","source_ids":["src-discovery-mims-harvard-tdc"],"source_locator":"Pinned README: data functions; dataset splits; evaluators; benchmark groups"},{"label":"Splits","value":"Random and scaffold-based splits are supported with recorded seeds/fractions; benchmark groups expose default split routines.","status":"source_checked","source_ids":["src-discovery-mims-harvard-tdc"],"source_locator":"Pinned README: data functions; dataset splits; evaluators; benchmark groups"},{"label":"Metrics","value":"A named evaluator computes task metrics; ROC-AUC is one documented example, not a universal TDC metric.","status":"source_checked","source_ids":["src-discovery-mims-harvard-tdc"],"source_locator":"Pinned README: data functions; dataset splits; evaluators; benchmark groups"},{"label":"Baselines","value":"The original molecular ADMET example compares RDKit2D-descriptor MLPs, Morgan-fingerprint MLPs and SMILES CNNs. Other in-scope molecular tasks require their own comparator set; these are not universal baselines for all of TDC.","status":"source_checked","source_ids":["evidence-discovery-final-tdc"],"source_locator":"Section 9 and Tables 3–4: benchmark groups"},{"label":"Leakage controls","value":"Scaffold holdout is available for relevant molecular tasks; it is not implied for every TDC dataset.","status":"source_checked","source_ids":["src-discovery-mims-harvard-tdc"],"source_locator":"Pinned README: data functions; dataset splits; evaluators; benchmark groups"},{"label":"Uncertainty","value":"The original ADMET table prints ± values but its caption and accompanying protocol do not define a suite-wide resampling or uncertainty rule. A result must retain the exact benchmark-group submission protocol before those values are interpreted.","status":"unreported","source_ids":["evidence-discovery-final-tdc"],"source_locator":"Section 9 and Tables 3–4: benchmark groups"},{"label":"Entity type","value":"Task and benchmark-group platform for molecular prediction.","status":"source_checked","source_ids":["src-discovery-mims-harvard-tdc"],"source_locator":"Pinned README: data functions; dataset splits; evaluators; benchmark groups"},{"label":"Organisms","value":"Organism scope is dataset-specific; the umbrella platform does not define one species.","status":"inapplicable","source_ids":["src-discovery-mims-harvard-tdc"],"source_locator":"Pinned README: data functions; dataset splits; evaluators; benchmark groups"},{"label":"Assays","value":"Task-specific molecular and therapeutic-property labels.","status":"source_checked","source_ids":["src-discovery-mims-harvard-tdc"],"source_locator":"Pinned README: data functions; dataset splits; evaluators; benchmark groups"},{"label":"Allowed inputs","value":"Named dataset inputs and labels; the benchmark group identifies the relevant molecular representation.","status":"source_checked","source_ids":["src-discovery-mims-harvard-tdc"],"source_locator":"Pinned README: data functions; dataset splits; evaluators; benchmark groups"},{"label":"Adaptation","value":"Dataset-specific supervised evaluation with default or explicitly selected split methods.","status":"source_checked","source_ids":["src-discovery-mims-harvard-tdc"],"source_locator":"Pinned README: data functions; dataset splits; evaluators; benchmark groups"}],"strengths":[{"text":"Benchmark groups expose standard split and evaluator routines while permitting explicit alternatives.","source_ids":["src-discovery-mims-harvard-tdc"],"source_locator":"Pinned README: data functions; dataset splits; evaluators; benchmark groups"}],"limitations":[{"text":"TDC also contains tasks outside this catalogue’s molecular remit. Shared software access does not make datasets, labels or evaluation protocols scientifically interchangeable.","source_ids":["evidence-discovery-final-tdc"],"source_locator":"Section 9 and Tables 3–4: benchmark groups"}],"diagram":{"title":"Evaluation procedure","steps":["Allowed inputs: Named dataset inputs and labels; the benchmark group identifies the relevant molecular representation.","Splits: Random and scaffold-based splits are supported with recorded seeds/fractions; benchmark groups expose default split routines.","Metrics: A named evaluator computes task metrics; ROC-AUC is one documented example, not a universal TDC metric."],"caption":"Conceptual procedure. Task variants and protocol versions retain their separate scoring conditions.","source_ids":["src-discovery-mims-harvard-tdc"],"source_locator":"Pinned README: data functions; dataset splits; evaluators; benchmark groups"},"coverage":"limited","gaps":["Uncertainty: The original ADMET table prints ± values but its caption and accompanying protocol do not define a suite-wide resampling or uncertainty rule. A result must retain the exact benchmark-group submission protocol before those values are interpreted."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary paper and/or task implementation reviewed for the explicitly cited methodology claims. Scope-limited absence is recorded only after the documented source search; no model runs or independent reproduction."}}},"description":"Molecular binding, biochemical activity and related specialist tasks","facets":{"areas":["molecular-interactions"]},"id":"discovery-benchmark-tdc-molecular-tasks","kind":"benchmark","links":[],"name":"TDC molecular tasks","source_ids":["src-discovery-mims-harvard-tdc"],"status":"discovered"} {"attributes":{"entity_level":"challenge","scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Zero-shot perturbation prediction in unseen cellular contexts","version":null,"historical_missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"The 2026 Virtual Cell Challenge evaluates perturbation-response prediction in unseen cellular contexts.","summary_source_ids":["evidence-benchmark-vcc2026-snapshot"],"summary_source_locator":"Arc Institute 2026 Virtual Cell Challenge announcement: task; context holdouts; inputs; scoring","sections":[{"title":"Evaluation methodology","body":"The 2026 challenge predicts CRISPRi responses in cellular contexts with no challenge-specific perturbation training set. Participants receive unperturbed cells and target-gene identifiers; three contexts support validation and three different contexts support final testing. Scores are calibrated against baseline and split-half experimental references.","source_ids":["evidence-discovery-final-vcc-guide"],"source_locator":"VCC CLI guide: context labels and interpretation of the six metrics"}],"facts":[{"label":"Datasets","value":"Challenge-specific cellular-context evaluation data; the previous year’s released dataset is a separate resource.","status":"source_checked","source_ids":["evidence-benchmark-vcc2026-snapshot"],"source_locator":"Arc Institute 2026 Virtual Cell Challenge announcement: task; context holdouts; inputs; scoring"},{"label":"Splits","value":"Validation contexts support the live leaderboard and other contexts are held for final testing; there is no challenge-specific training set.","status":"source_checked","source_ids":["evidence-benchmark-vcc2026-snapshot"],"source_locator":"Arc Institute 2026 Virtual Cell Challenge announcement: task; context holdouts; inputs; scoring"},{"label":"Metrics","value":"Six calibrated components: perturbation discrimination, expression accuracy, differential-expression log-fold-change accuracy, direction fidelity, direction reach and significance overlap. Their unweighted mean is the overall score; each result also needs its partition, perturbation panel and anchor-set identity.","status":"source_checked","source_ids":["evidence-discovery-final-vcc-guide"],"source_locator":"VCC CLI guide: score interpretation and partition/panel/anchor stamp"},{"label":"Baselines","value":"Participants may choose modeling strategies and train on public or their own datasets.","status":"source_checked","source_ids":["evidence-benchmark-vcc2026-snapshot"],"source_locator":"Arc Institute 2026 Virtual Cell Challenge announcement: task; context holdouts; inputs; scoring"},{"label":"Leakage controls","value":"Unseen-context generalization defines the challenge; exact admissibility and overlap rules require the detailed competition protocol.","status":"source_checked","source_ids":["evidence-benchmark-vcc2026-snapshot"],"source_locator":"Arc Institute 2026 Virtual Cell Challenge announcement: task; context holdouts; inputs; scoring"},{"label":"Uncertainty","value":"The score scale uses a mean-effect baseline and a split-half replicate of the real experiment as reference points. Replicate noise is therefore represented in calibration, but the guide does not define a universal confidence interval on the final leaderboard score.","status":"source_checked","source_ids":["evidence-discovery-final-vcc-guide"],"source_locator":"VCC CLI guide: context labels and interpretation of the six metrics"},{"label":"Entity type","value":"Dated cellular perturbation prediction challenge.","status":"source_checked","source_ids":["evidence-benchmark-vcc2026-snapshot"],"source_locator":"Arc Institute 2026 Virtual Cell Challenge announcement: task; context holdouts; inputs; scoring"},{"label":"Organisms","value":"The released contexts are deliberately anonymized cell lines A/B/C for validation and D/E/F for final testing. The inspected public announcement and CLI guide do not provide their exact line identities or a per-context organism manifest; do not infer them from the 2025 H1 dataset.","status":"unreported","source_ids":["evidence-discovery-final-vcc-guide"],"source_locator":"VCC CLI guide: context labels and interpretation of the six metrics"},{"label":"Assays","value":"Challenge-specific perturbation response data; prior-year data form a different resource.","status":"source_checked","source_ids":["evidence-benchmark-vcc2026-snapshot"],"source_locator":"Arc Institute 2026 Virtual Cell Challenge announcement: task; context holdouts; inputs; scoring"},{"label":"Allowed inputs","value":"Released challenge inputs and the submission schema for the 2026 edition.","status":"source_checked","source_ids":["evidence-benchmark-vcc2026-snapshot"],"source_locator":"Arc Institute 2026 Virtual Cell Challenge announcement: task; context holdouts; inputs; scoring"},{"label":"Adaptation","value":"Rules and deadlines belong to this challenge edition; prior-year adaptation conditions are not automatically inherited.","status":"source_checked","source_ids":["evidence-benchmark-vcc2026-snapshot"],"source_locator":"Arc Institute 2026 Virtual Cell Challenge announcement: task; context holdouts; inputs; scoring"}],"strengths":[{"text":"A dated challenge identity prevents historical datasets being mistaken for the current evaluation.","source_ids":["evidence-benchmark-vcc2026-snapshot"],"source_locator":"Arc Institute 2026 Virtual Cell Challenge announcement: task; context holdouts; inputs; scoring"}],"limitations":[{"text":"Validation and final phases use different contexts, perturbation panels and scoring bundles. Their scores are not directly comparable, and a replicate-calibrated value is neither a percentage correct nor an absolute biological ceiling.","source_ids":["evidence-discovery-final-vcc-guide"],"source_locator":"VCC CLI guide: context labels and interpretation of the six metrics"}],"diagram":{"title":"Evaluation procedure","steps":["Allowed inputs: Released challenge inputs and the submission schema for the 2026 edition.","Splits: Validation contexts support the live leaderboard and other contexts are held for final testing; there is no challenge-specific training set.","Metrics: Six calibrated components: perturbation discrimination, expression accuracy, differential-expression log-fold-change accuracy, direction fidelity, direction reach and significance overlap. Their unweighted mean is the overall score; each result also needs its partition, perturbation panel and anchor-set identity."],"caption":"Conceptual procedure. Task variants and protocol versions retain their separate scoring conditions.","source_ids":["evidence-benchmark-vcc2026-snapshot","evidence-discovery-final-vcc-guide"],"source_locator":"Arc Institute 2026 Virtual Cell Challenge announcement: task; context holdouts; inputs; scoring; VCC CLI guide: score interpretation and partition/panel/anchor stamp"},"coverage":"limited","gaps":["Organisms: The released contexts are deliberately anonymized cell lines A/B/C for validation and D/E/F for final testing. The inspected public announcement and CLI guide do not provide their exact line identities or a per-context organism manifest; do not infer them from the 2025 H1 dataset."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary paper and/or task implementation reviewed for the explicitly cited methodology claims. Scope-limited absence is recorded only after the documented source search; no model runs or independent reproduction."}}},"description":"Zero-shot perturbation prediction in unseen cellular contexts","facets":{"areas":["single-cell"]},"id":"discovery-benchmark-virtual-cell-challenge-2026","kind":"benchmark","links":[],"name":"Virtual Cell Challenge 2026","source_ids":["src-discovery-vcc2026"],"status":"discovered"} {"attributes":{"assay":"mass spectrometry","missing_metadata":{"split":"not_applicable","version":"unextracted"},"scope_note":"Reference library; a leakage-aware benchmark split and scoring protocol must be defined separately.","split":null,"version":null},"description":"Experimental lipid reference spectra for identification assessment.","facets":{"areas":["lipidomics"]},"id":"discovery-dataset-lipid-maps-standards-spectra","kind":"dataset","links":[],"name":"LIPID MAPS Standards Spectra","source_ids":["src-discovery-lipidmaps-spectra"],"status":"discovered"} {"attributes":{"missing_metadata":{"denominator":"unextracted","split":"unextracted","version":"unreported"},"split":null,"version":null},"description":"Dataset used by the historical TAPE leaderboard.","facets":{"areas":["protein-function"]},"id":"discovery-dataset-tape-fluorescence-source-dataset","kind":"dataset","links":[],"name":"TAPE Fluorescence source dataset","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"missing_metadata":{"denominator":"unextracted","split":"unextracted","version":"unreported"},"split":null,"version":null},"description":"Dataset used by the historical TAPE leaderboard.","facets":{"areas":["protein-function"]},"id":"discovery-dataset-tape-stability-source-dataset","kind":"dataset","links":[],"name":"TAPE Stability source dataset","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"comparison":{"adaptation":null,"aggregation":null,"budget":null,"dataset_version":null,"inputs":null,"metric_implementation":null,"population":null,"protocol_id":"discovery-benchmark-tape-fluorescence","split":null},"missing_metadata":{"checkpoint":"unreported","denominator":"unextracted","seeds":"unreported","split_manifest":"unextracted"},"origin":"author_reported","protocol":"TAPE Fluorescence","version":"6d345c2b2bbf52cd32cf179325c222afd92aec7e"},"description":"Historical benchmark-maintainer report; experiments have not been rerun.","facets":{"areas":["protein-function"]},"id":"discovery-evaluation-tape-fluorescence-bepler-leaderboard-evaluation","kind":"evaluation","links":[{"relation":"model","target_id":"discovery-model-tape-bepler"},{"relation":"benchmark","target_id":"discovery-benchmark-tape-fluorescence"},{"relation":"dataset","target_id":"discovery-dataset-tape-fluorescence-source-dataset"}],"name":"TAPE Fluorescence Bepler leaderboard evaluation","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"comparison":{"adaptation":null,"aggregation":null,"budget":null,"dataset_version":null,"inputs":null,"metric_implementation":null,"population":null,"protocol_id":"discovery-benchmark-tape-fluorescence","split":null},"missing_metadata":{"checkpoint":"unreported","denominator":"unextracted","seeds":"unreported","split_manifest":"unextracted"},"origin":"author_reported","protocol":"TAPE Fluorescence","version":"6d345c2b2bbf52cd32cf179325c222afd92aec7e"},"description":"Historical benchmark-maintainer report; experiments have not been rerun.","facets":{"areas":["protein-function"]},"id":"discovery-evaluation-tape-fluorescence-lstm-leaderboard-evaluation","kind":"evaluation","links":[{"relation":"model","target_id":"discovery-model-tape-lstm"},{"relation":"benchmark","target_id":"discovery-benchmark-tape-fluorescence"},{"relation":"dataset","target_id":"discovery-dataset-tape-fluorescence-source-dataset"}],"name":"TAPE Fluorescence LSTM leaderboard evaluation","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"comparison":{"adaptation":null,"aggregation":null,"budget":null,"dataset_version":null,"inputs":null,"metric_implementation":null,"population":null,"protocol_id":"discovery-benchmark-tape-fluorescence","split":null},"missing_metadata":{"checkpoint":"unreported","denominator":"unextracted","seeds":"unreported","split_manifest":"unextracted"},"origin":"author_reported","protocol":"TAPE Fluorescence","version":"6d345c2b2bbf52cd32cf179325c222afd92aec7e"},"description":"Historical benchmark-maintainer report; experiments have not been rerun.","facets":{"areas":["protein-function"]},"id":"discovery-evaluation-tape-fluorescence-one-hot-leaderboard-evaluation","kind":"evaluation","links":[{"relation":"model","target_id":"discovery-model-tape-one-hot"},{"relation":"benchmark","target_id":"discovery-benchmark-tape-fluorescence"},{"relation":"dataset","target_id":"discovery-dataset-tape-fluorescence-source-dataset"}],"name":"TAPE Fluorescence One Hot leaderboard evaluation","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"comparison":{"adaptation":null,"aggregation":null,"budget":null,"dataset_version":null,"inputs":null,"metric_implementation":null,"population":null,"protocol_id":"discovery-benchmark-tape-fluorescence","split":null},"missing_metadata":{"checkpoint":"unreported","denominator":"unextracted","seeds":"unreported","split_manifest":"unextracted"},"origin":"author_reported","protocol":"TAPE Fluorescence","version":"6d345c2b2bbf52cd32cf179325c222afd92aec7e"},"description":"Historical benchmark-maintainer report; experiments have not been rerun.","facets":{"areas":["protein-function"]},"id":"discovery-evaluation-tape-fluorescence-resnet-leaderboard-evaluation","kind":"evaluation","links":[{"relation":"model","target_id":"discovery-model-tape-resnet"},{"relation":"benchmark","target_id":"discovery-benchmark-tape-fluorescence"},{"relation":"dataset","target_id":"discovery-dataset-tape-fluorescence-source-dataset"}],"name":"TAPE Fluorescence ResNet leaderboard evaluation","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"comparison":{"adaptation":null,"aggregation":null,"budget":null,"dataset_version":null,"inputs":null,"metric_implementation":null,"population":null,"protocol_id":"discovery-benchmark-tape-fluorescence","split":null},"missing_metadata":{"checkpoint":"unreported","denominator":"unextracted","seeds":"unreported","split_manifest":"unextracted"},"origin":"author_reported","protocol":"TAPE Fluorescence","version":"6d345c2b2bbf52cd32cf179325c222afd92aec7e"},"description":"Historical benchmark-maintainer report; experiments have not been rerun.","facets":{"areas":["protein-function"]},"id":"discovery-evaluation-tape-fluorescence-transformer-leaderboard-evaluation","kind":"evaluation","links":[{"relation":"model","target_id":"discovery-model-tape-transformer"},{"relation":"benchmark","target_id":"discovery-benchmark-tape-fluorescence"},{"relation":"dataset","target_id":"discovery-dataset-tape-fluorescence-source-dataset"}],"name":"TAPE Fluorescence Transformer leaderboard evaluation","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"comparison":{"adaptation":null,"aggregation":null,"budget":null,"dataset_version":null,"inputs":null,"metric_implementation":null,"population":null,"protocol_id":"discovery-benchmark-tape-fluorescence","split":null},"missing_metadata":{"checkpoint":"unreported","denominator":"unextracted","seeds":"unreported","split_manifest":"unextracted"},"origin":"author_reported","protocol":"TAPE Fluorescence","version":"6d345c2b2bbf52cd32cf179325c222afd92aec7e"},"description":"Historical benchmark-maintainer report; experiments have not been rerun.","facets":{"areas":["protein-function"]},"id":"discovery-evaluation-tape-fluorescence-unirep-leaderboard-evaluation","kind":"evaluation","links":[{"relation":"model","target_id":"discovery-model-tape-unirep"},{"relation":"benchmark","target_id":"discovery-benchmark-tape-fluorescence"},{"relation":"dataset","target_id":"discovery-dataset-tape-fluorescence-source-dataset"}],"name":"TAPE Fluorescence Unirep leaderboard evaluation","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"comparison":{"adaptation":null,"aggregation":null,"budget":null,"dataset_version":null,"inputs":null,"metric_implementation":null,"population":null,"protocol_id":"discovery-benchmark-tape-stability","split":null},"missing_metadata":{"checkpoint":"unreported","denominator":"unextracted","seeds":"unreported","split_manifest":"unextracted"},"origin":"author_reported","protocol":"TAPE Stability","version":"6d345c2b2bbf52cd32cf179325c222afd92aec7e"},"description":"Historical benchmark-maintainer report; experiments have not been rerun.","facets":{"areas":["protein-function"]},"id":"discovery-evaluation-tape-stability-bepler-leaderboard-evaluation","kind":"evaluation","links":[{"relation":"model","target_id":"discovery-model-tape-bepler"},{"relation":"benchmark","target_id":"discovery-benchmark-tape-stability"},{"relation":"dataset","target_id":"discovery-dataset-tape-stability-source-dataset"}],"name":"TAPE Stability Bepler leaderboard evaluation","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"comparison":{"adaptation":null,"aggregation":null,"budget":null,"dataset_version":null,"inputs":null,"metric_implementation":null,"population":null,"protocol_id":"discovery-benchmark-tape-stability","split":null},"missing_metadata":{"checkpoint":"unreported","denominator":"unextracted","seeds":"unreported","split_manifest":"unextracted"},"origin":"author_reported","protocol":"TAPE Stability","version":"6d345c2b2bbf52cd32cf179325c222afd92aec7e"},"description":"Historical benchmark-maintainer report; experiments have not been rerun.","facets":{"areas":["protein-function"]},"id":"discovery-evaluation-tape-stability-lstm-leaderboard-evaluation","kind":"evaluation","links":[{"relation":"model","target_id":"discovery-model-tape-lstm"},{"relation":"benchmark","target_id":"discovery-benchmark-tape-stability"},{"relation":"dataset","target_id":"discovery-dataset-tape-stability-source-dataset"}],"name":"TAPE Stability LSTM leaderboard evaluation","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"comparison":{"adaptation":null,"aggregation":null,"budget":null,"dataset_version":null,"inputs":null,"metric_implementation":null,"population":null,"protocol_id":"discovery-benchmark-tape-stability","split":null},"missing_metadata":{"checkpoint":"unreported","denominator":"unextracted","seeds":"unreported","split_manifest":"unextracted"},"origin":"author_reported","protocol":"TAPE Stability","version":"6d345c2b2bbf52cd32cf179325c222afd92aec7e"},"description":"Historical benchmark-maintainer report; experiments have not been rerun.","facets":{"areas":["protein-function"]},"id":"discovery-evaluation-tape-stability-one-hot-leaderboard-evaluation","kind":"evaluation","links":[{"relation":"model","target_id":"discovery-model-tape-one-hot"},{"relation":"benchmark","target_id":"discovery-benchmark-tape-stability"},{"relation":"dataset","target_id":"discovery-dataset-tape-stability-source-dataset"}],"name":"TAPE Stability One Hot leaderboard evaluation","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"comparison":{"adaptation":null,"aggregation":null,"budget":null,"dataset_version":null,"inputs":null,"metric_implementation":null,"population":null,"protocol_id":"discovery-benchmark-tape-stability","split":null},"missing_metadata":{"checkpoint":"unreported","denominator":"unextracted","seeds":"unreported","split_manifest":"unextracted"},"origin":"author_reported","protocol":"TAPE Stability","version":"6d345c2b2bbf52cd32cf179325c222afd92aec7e"},"description":"Historical benchmark-maintainer report; experiments have not been rerun.","facets":{"areas":["protein-function"]},"id":"discovery-evaluation-tape-stability-resnet-leaderboard-evaluation","kind":"evaluation","links":[{"relation":"model","target_id":"discovery-model-tape-resnet"},{"relation":"benchmark","target_id":"discovery-benchmark-tape-stability"},{"relation":"dataset","target_id":"discovery-dataset-tape-stability-source-dataset"}],"name":"TAPE Stability ResNet leaderboard evaluation","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"comparison":{"adaptation":null,"aggregation":null,"budget":null,"dataset_version":null,"inputs":null,"metric_implementation":null,"population":null,"protocol_id":"discovery-benchmark-tape-stability","split":null},"missing_metadata":{"checkpoint":"unreported","denominator":"unextracted","seeds":"unreported","split_manifest":"unextracted"},"origin":"author_reported","protocol":"TAPE Stability","version":"6d345c2b2bbf52cd32cf179325c222afd92aec7e"},"description":"Historical benchmark-maintainer report; experiments have not been rerun.","facets":{"areas":["protein-function"]},"id":"discovery-evaluation-tape-stability-transformer-leaderboard-evaluation","kind":"evaluation","links":[{"relation":"model","target_id":"discovery-model-tape-transformer"},{"relation":"benchmark","target_id":"discovery-benchmark-tape-stability"},{"relation":"dataset","target_id":"discovery-dataset-tape-stability-source-dataset"}],"name":"TAPE Stability Transformer leaderboard evaluation","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"comparison":{"adaptation":null,"aggregation":null,"budget":null,"dataset_version":null,"inputs":null,"metric_implementation":null,"population":null,"protocol_id":"discovery-benchmark-tape-stability","split":null},"missing_metadata":{"checkpoint":"unreported","denominator":"unextracted","seeds":"unreported","split_manifest":"unextracted"},"origin":"author_reported","protocol":"TAPE Stability","version":"6d345c2b2bbf52cd32cf179325c222afd92aec7e"},"description":"Historical benchmark-maintainer report; experiments have not been rerun.","facets":{"areas":["protein-function"]},"id":"discovery-evaluation-tape-stability-unirep-leaderboard-evaluation","kind":"evaluation","links":[{"relation":"model","target_id":"discovery-model-tape-unirep"},{"relation":"benchmark","target_id":"discovery-benchmark-tape-stability"},{"relation":"dataset","target_id":"discovery-dataset-tape-stability-source-dataset"}],"name":"TAPE Stability Unirep leaderboard evaluation","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-geneb"],"entity_level":"family","reported_name":"Agro Nucleotide Transformer","version":null,"historical_missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"AgroNT learns DNA representations from plant reference genomes for plant molecular prediction tasks.","summary_source_ids":["evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-4603e2d255507e025c80","evidence-official-04ac37076f906b8470cb","evidence-official-576c2ecba240ddfda8f2"],"summary_source_locator":"AgroNT paper Methods: Architecture, Pre-training dataset and Pre-training strategy; official model-card licence metadata","sections":[{"title":"How it works","body":"AgroNT learns DNA representations from plant reference genomes for plant molecular prediction tasks. One-billion-parameter encoder-only transformer with 40 attention blocks, hidden width 1,500, learned positional embeddings and a six-mer masked-language-model head. The documented inputs are plant DNA sequence, with standalone tokens for ambiguous or remainder bases. The output consists of DNA embeddings for downstream regulatory, RNA-processing or expression tasks.","source_ids":["evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-4603e2d255507e025c80","evidence-official-04ac37076f906b8470cb","evidence-official-576c2ecba240ddfda8f2"],"source_locator":"AgroNT paper Methods: Architecture, Pre-training dataset and Pre-training strategy; official model-card licence metadata"},{"title":"Versions and reproducibility","body":"1B_agro_nt pretrained model. 1,024 tokens, approximately 6kb of unambiguous sequence rather than an unconditional 6,144-base guarantee.","source_ids":["evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-4603e2d255507e025c80","evidence-official-04ac37076f906b8470cb","evidence-official-576c2ecba240ddfda8f2"],"source_locator":"AgroNT paper Methods: Architecture, Pre-training dataset and Pre-training strategy; official model-card licence metadata"}],"facts":[{"label":"Model type","value":"Plant DNA transformer encoder","status":"source_checked","source_ids":["evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-4603e2d255507e025c80","evidence-official-04ac37076f906b8470cb","evidence-official-576c2ecba240ddfda8f2"],"source_locator":"AgroNT paper Methods: Architecture, Pre-training dataset and Pre-training strategy; official model-card licence metadata"},{"label":"Architecture","value":"One-billion-parameter encoder-only transformer with 40 attention blocks, hidden width 1,500, learned positional embeddings and a six-mer masked-language-model head.","status":"source_checked","source_ids":["evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-4603e2d255507e025c80","evidence-official-04ac37076f906b8470cb","evidence-official-576c2ecba240ddfda8f2"],"source_locator":"AgroNT paper Methods: Architecture, Pre-training dataset and Pre-training strategy; official model-card licence metadata"},{"label":"Inputs","value":"Plant DNA sequence, with standalone tokens for ambiguous or remainder bases.","status":"source_checked","source_ids":["evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-4603e2d255507e025c80","evidence-official-04ac37076f906b8470cb","evidence-official-576c2ecba240ddfda8f2"],"source_locator":"AgroNT paper Methods: Architecture, Pre-training dataset and Pre-training strategy; official model-card licence metadata"},{"label":"Outputs","value":"DNA embeddings for downstream regulatory, RNA-processing or expression tasks.","status":"source_checked","source_ids":["evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-4603e2d255507e025c80","evidence-official-04ac37076f906b8470cb","evidence-official-576c2ecba240ddfda8f2"],"source_locator":"AgroNT paper Methods: Architecture, Pre-training dataset and Pre-training strategy; official model-card licence metadata"},{"label":"Parameters","value":"1 billion.","status":"source_checked","source_ids":["evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-4603e2d255507e025c80","evidence-official-04ac37076f906b8470cb","evidence-official-576c2ecba240ddfda8f2"],"source_locator":"AgroNT paper Methods: Architecture, Pre-training dataset and Pre-training strategy; official model-card licence metadata"},{"label":"Known versions","value":"1B_agro_nt pretrained model.","status":"source_checked","source_ids":["evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-4603e2d255507e025c80","evidence-official-04ac37076f906b8470cb","evidence-official-576c2ecba240ddfda8f2"],"source_locator":"AgroNT paper Methods: Architecture, Pre-training dataset and Pre-training strategy; official model-card licence metadata"},{"label":"Training data","value":"Approximately 10.5M sequences from reference genomes of 48 plant species in Ensembl Plants.","status":"source_checked","source_ids":["evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-4603e2d255507e025c80","evidence-official-04ac37076f906b8470cb","evidence-official-576c2ecba240ddfda8f2"],"source_locator":"AgroNT paper Methods: Architecture, Pre-training dataset and Pre-training strategy; official model-card licence metadata"},{"label":"Training cutoff","value":"The inspected Methods identifies 48 Ensembl Plants reference species. It does not state one latest-deposition date for their combined genomic sequences.","status":"unreported","source_ids":["evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-4603e2d255507e025c80","evidence-official-04ac37076f906b8470cb","evidence-official-576c2ecba240ddfda8f2"],"source_locator":"AgroNT paper Methods: Architecture, Pre-training dataset and Pre-training strategy; official model-card licence metadata"},{"label":"Context limits","value":"1,024 tokens, approximately 6kb of unambiguous sequence rather than an unconditional 6,144-base guarantee.","status":"source_checked","source_ids":["evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-4603e2d255507e025c80","evidence-official-04ac37076f906b8470cb","evidence-official-576c2ecba240ddfda8f2"],"source_locator":"AgroNT paper Methods: Architecture, Pre-training dataset and Pre-training strategy; official model-card licence metadata"},{"label":"Weights licence","value":"CC-BY-NC-SA-4.0 declared by the official agro-nucleotide-transformer-1b model card.","status":"source_checked","source_ids":["evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-4603e2d255507e025c80","evidence-official-04ac37076f906b8470cb","evidence-official-576c2ecba240ddfda8f2"],"source_locator":"AgroNT paper Methods: Architecture, Pre-training dataset and Pre-training strategy; official model-card licence metadata"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/instadeepai/nucleotide-transformer","status":"source_checked","source_ids":["evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-4603e2d255507e025c80","evidence-official-04ac37076f906b8470cb","evidence-official-576c2ecba240ddfda8f2"],"source_locator":"AgroNT paper Methods: Architecture, Pre-training dataset and Pre-training strategy; official model-card licence metadata"},{"label":"Code licence","value":"CC-BY-NC-SA-4.0","status":"source_checked","source_ids":["evidence-official-7e4b193e47ba209860a1"],"source_locator":"LICENSE.md: licence text"}],"strengths":[{"text":"Pretraining explicitly targets plant reference genomes, predominantly crop species.","source_ids":["evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-4603e2d255507e025c80","evidence-official-04ac37076f906b8470cb","evidence-official-576c2ecba240ddfda8f2"],"source_locator":"AgroNT paper Methods: Architecture, Pre-training dataset and Pre-training strategy; official model-card licence metadata"}],"limitations":[{"text":"The context is counted in tokens: the class token and ambiguous bases reduce the maximum number of ordinary bases represented. Predictions outside the studied species/tasks need separate evaluation.","source_ids":["evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-4603e2d255507e025c80","evidence-official-04ac37076f906b8470cb","evidence-official-576c2ecba240ddfda8f2"],"source_locator":"AgroNT paper Methods: Architecture, Pre-training dataset and Pre-training strategy; official model-card licence metadata"}],"diagram":{"title":"Agro Nucleotide Transformer workflow","steps":["Plant DNA","6-mer tokenizer","Masked-language transformer","Plant DNA embeddings"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-4603e2d255507e025c80","evidence-official-04ac37076f906b8470cb","evidence-official-576c2ecba240ddfda8f2"],"source_locator":"AgroNT paper Methods: Architecture, Pre-training dataset and Pre-training strategy; official model-card licence metadata"},"coverage":"limited","gaps":["Training cutoff: The inspected Methods identifies 48 Ensembl Plants reference species. It does not state one latest-deposition date for their combined genomic sequences."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}},"description":"Plant genomic representation family","facets":{"areas":["genomics"]},"id":"discovery-model-agro-nucleotide-transformer","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-geneb"}],"name":"Agro Nucleotide Transformer","source_ids":["src-discovery-instadeepai-nucleotide-transformer"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-plinder"],"entity_level":"family","reported_name":"AlphaFold 3","version":null,"historical_missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"AlphaFold 3 predicts three-dimensional structures of complexes containing proteins, nucleic acids and other molecular components. It combines a Pairformer representation network with an atomic-coordinate diffusion model. This entry describes the model and local implementation; the hosted AlphaFold Server has a separate profile.","summary_source_ids":["evidence-alphafold-paper","evidence-alphafold-readme"],"summary_source_locator":"Abstract; Model architecture; README: Installation and Usage","sections":[{"title":"Joint structure prediction","body":"Sequence, chemical and evolutionary features feed a Pairformer, which builds representations of individual tokens and their relationships. A diffusion module then predicts atomic coordinates. Separate heads estimate confidence. The paper describes 48 Pairformer blocks; the architecture models complexes jointly rather than treating every partner as a separately folded structure.","source_ids":["evidence-alphafold-paper"],"source_locator":"Main text: Model architecture; Fig. 1d and Fig. 2"},{"title":"Training and evaluation context","body":"The standard model uses a structural training cutoff of 30 September 2021. The dedicated PoseBusters Methods section reports a separate model with a 30 September 2019 cutoff, although other training passages disagree (see limitations). Model seeds, templates, input information and ranking also affect the reported comparison; paper evaluation variants are not automatically identical to current downloadable weights.","source_ids":["evidence-alphafold-paper","evidence-alphafold-supplement"],"source_locator":"Main paper Methods: Training regime, Inference regime and PoseBusters; supplement Section 5.2"},{"title":"Local access","body":"The pinned repository provides inference code and a direct Google-hosted weights download. Its README is more current on access than the server FAQ, which still describes an application form. Code and model parameters have different licences; the server output terms should not be substituted for the local weights terms.","source_ids":["evidence-alphafold-readme","evidence-alphafold-license","evidence-alphafold-weights-terms-of-use","evidence-alphafold-server-faq"],"source_locator":"README: Obtaining Model Parameters and Licences; LICENSE; weights terms: Key things to know; FAQ: model access"},{"title":"Training data and provenance","body":"Training combines experimental PDB structures with approximately 41 million predicted protein monomers, about 25,000 disorder-focused protein complexes and about 65,000 predicted RNA structures. The supplement also lists transcription-factor examples used during fine-tuning. Its sequence-search resources include UniRef90, UniProt, BFD/Uniclust30, MGnify, Rfam and RNAcentral. These resources have different versions and dates: the structural training cutoff is not a cutoff for every sequence database.","source_ids":["evidence-alphafold-supplement"],"source_locator":"Sections 2.2 and 2.5; Table 3; Section 2.5.2 distillation datasets (PDF pages 8–9, printed pages 3–4)"}],"facts":[{"label":"Model type","value":"Pairformer plus diffusion model for joint biomolecular structure prediction.","status":"source_checked","source_ids":["evidence-alphafold-paper"],"source_locator":"Model architecture; Fig. 1"},{"label":"Architecture","value":"A 48-block Pairformer builds token and pair representations; a diffusion module predicts atomic coordinates and separate heads estimate confidence.","status":"source_checked","source_ids":["evidence-alphafold-paper"],"source_locator":"Model architecture; Fig. 1d and Fig. 2"},{"label":"Known versions","value":"AlphaFold 3; inference code/documentation reviewed at commit c0f97eda2f1f482fd94d3a38bece18c7069b4a5c. This is a software revision, not a weight-file checksum. Paper evaluation variants are distinct configurations.","status":"source_checked","source_ids":["evidence-alphafold-readme","evidence-alphafold-paper"],"source_locator":"Pinned repository revision; paper evaluation distinctions described in Methods"},{"label":"Inputs","value":"Protein, DNA and RNA sequences; chemical components; optional MSAs and structural templates.","status":"source_checked","source_ids":["evidence-alphafold-docs-input"],"source_locator":"Top-level structure; protein, RNA, DNA and ligand inputs"},{"label":"Outputs","value":"Predicted structures in mmCIF plus confidence outputs, including pLDDT, PAE, pTM and ipTM.","status":"source_checked","source_ids":["evidence-alphafold-docs-output"],"source_locator":"Output directory structure; confidence outputs"},{"label":"Training cutoff","value":"Experimental PDB structures plus protein and RNA distillation sets. The standard structural cutoff is 2021-09-30. The dedicated PoseBusters Methods section specifies a separate 2019-09-30 model; other training passages conflict with that date (see below).","status":"source_checked","source_ids":["evidence-alphafold-paper","evidence-alphafold-supplement"],"source_locator":"Main paper Methods: Training regime and PoseBusters; Supplement Sections 2.5 and 5.2"},{"label":"Training data","value":"Experimental PDB structures, protein and RNA distillation sets, and transcription-factor examples used during fine-tuning. Sequence-search databases are separately versioned input resources.","status":"source_checked","source_ids":["evidence-alphafold-supplement"],"source_locator":"Sections 2.2 and 2.5; Table 3; training-data discussion below"},{"label":"Context limits","value":"The default largest compilation bucket is 5,120 tokens. The documentation supports larger inputs by configuration, subject to memory; this is not a universal architectural context limit.","status":"source_checked","source_ids":["evidence-alphafold-docs-performance"],"source_locator":"Compilation buckets; predicting structures with more than 5,120 tokens"},{"label":"Access","value":"Public inference implementation; weights downloaded directly from Google under separate non-commercial terms.","status":"source_checked","source_ids":["evidence-alphafold-readme"],"source_locator":"Obtaining Model Parameters; Installation and Usage"},{"label":"Code licence","value":"Apache License 2.0.","status":"source_checked","source_ids":["evidence-alphafold-license"],"source_locator":"LICENSE"},{"label":"Weights licence","value":"Custom AlphaFold 3 Model Parameters Terms of Use, last modified 2024-11-09. Non-commercial use by or for non-commercial organisations; additional output and redistribution restrictions apply.","status":"source_checked","source_ids":["evidence-alphafold-weights-terms-of-use"],"source_locator":"Key things to know; Use restrictions"},{"label":"Parameters","value":"No total trainable-parameter count is reported in the inspected main paper, supplementary architecture/training sections or implementation documentation. Layer dimensions do not establish a complete checkpoint total.","status":"unreported","source_ids":["evidence-alphafold-paper","evidence-alphafold-supplement"],"source_locator":"Main paper Model architecture; supplement Sections 3–5 and full-text parameter search; implementation documentation"}],"strengths":[{"text":"One architecture handles several molecular component types and their joint structures. This is a capability description, not evidence that every complex will be accurate.","source_ids":["evidence-alphafold-paper"],"source_locator":"Abstract; Model architecture"},{"text":"The local implementation exposes input, template and output specifications, allowing an evaluation configuration to be documented.","source_ids":["evidence-alphafold-docs-input"],"source_locator":"Input format and optional input fields"}],"limitations":[{"text":"Predictions can contain incorrect chirality, atomic clashes or spurious structure in disordered regions. Confidence and structural plausibility need separate inspection.","source_ids":["evidence-alphafold-paper"],"source_locator":"Model limitations; Fig. 5"},{"text":"Sampled structures are not a calibrated solution-state ensemble. Prediction confidence does not establish binding affinity or experimental function.","source_ids":["evidence-alphafold-paper"],"source_locator":"Model limitations: dynamics and conformational states; confidence outputs are structure-quality estimates"},{"text":"The public code licence does not remove the separate non-commercial restrictions on weights and outputs.","source_ids":["evidence-alphafold-weights-terms-of-use"],"source_locator":"Key things to know"},{"text":"The source is internally inconsistent about the PoseBusters training cutoff. Its dedicated PoseBusters Methods section and Results specify 2019-09-30, while the general Training regime and supplement Section 5.2 say 2021-09-30. The separate evaluation variant is retained; this profile does not resolve the discrepancy or assign its result to a current downloadable checkpoint.","source_ids":["evidence-alphafold-paper","evidence-alphafold-supplement"],"source_locator":"Main paper Results and Methods: PoseBusters versus Training regime; supplement Section 5.2, printed page 29"}],"diagram":{"title":"AlphaFold 3 architecture","steps":["Molecular sequences and chemical features","MSA and template features","Pairformer token and pair representations","Diffusion predicts atomic coordinates","Confidence heads and ranked structures"],"caption":"Conceptual architecture based on the paper. Exact preprocessing and sampling settings belong to each evaluation.","source_ids":["evidence-alphafold-paper"],"source_locator":"Fig. 1d; Fig. 2; Model architecture"},"coverage":"reviewed","gaps":["The inspected sources do not report a complete checkpoint parameter total; no weights were downloaded or counted.","The PoseBusters cutoff disagreement is preserved explicitly. The currently downloaded weights have not been mapped to the paper’s separate evaluation variants."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Read primary paper XML, pinned official repository documentation and licences. Reviewed public server FAQ separately. No model run, independent performance replication or human review. Supplementary PDF reviewed, including visual checks of Tables 3 and 6. Conflicting cutoff statements remain explicit. A second automated reviewer checked the AlphaFold source claims and service/model distinction; this is not human review or experimental reproduction."}}},"description":"Biomolecular complex structure prediction","facets":{"areas":["protein-structure"]},"id":"discovery-model-alphafold-3","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-plinder"}],"name":"AlphaFold 3","source_ids":["src-discovery-google-deepmind-alphafold3"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-posebusters"],"entity_level":"method","reported_name":"AutoDock Vina","version":null,"historical_missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"AutoDock Vina searches for ligand conformations and poses in a molecular docking problem.","summary_source_ids":["evidence-official-23e72f7e93a0a633ab2b"],"summary_source_locator":"README.md: introduction, feature list, license and Citations","sections":[{"title":"How it works","body":"AutoDock Vina searches for ligand conformations and poses in a molecular docking problem. Scoring functions coupled to gradient-based conformational optimization and search. The documented inputs are prepared receptor and ligand structures with the configured search space and scoring function. The output consists of candidate docking poses and corresponding docking scores.","source_ids":["evidence-official-23e72f7e93a0a633ab2b"],"source_locator":"README.md: introduction, feature list, license and Citations"},{"title":"Versions and reproducibility","body":"AutoDock Vina; README cites the 1.2.0 feature expansion separately from the original 2010 method. Molecular geometry and search-box constraints rather than a sequence context window.","source_ids":["evidence-official-23e72f7e93a0a633ab2b"],"source_locator":"README.md: introduction, feature list, license and Citations"}],"facts":[{"label":"Model type","value":"Classical molecular docking software","status":"source_checked","source_ids":["evidence-official-23e72f7e93a0a633ab2b"],"source_locator":"README.md: introduction, feature list, license and Citations"},{"label":"Architecture","value":"Scoring functions coupled to gradient-based conformational optimization and search.","status":"source_checked","source_ids":["evidence-official-23e72f7e93a0a633ab2b"],"source_locator":"README.md: introduction, feature list, license and Citations"},{"label":"Inputs","value":"Prepared receptor and ligand structures with the configured search space and scoring function.","status":"source_checked","source_ids":["evidence-official-23e72f7e93a0a633ab2b"],"source_locator":"README.md: introduction, feature list, license and Citations"},{"label":"Outputs","value":"Candidate docking poses and corresponding docking scores.","status":"source_checked","source_ids":["evidence-official-23e72f7e93a0a633ab2b"],"source_locator":"README.md: introduction, feature list, license and Citations"},{"label":"Parameters","value":"Inapplicable as a neural parameter count; scoring/search parameters are separately configured.","status":"inapplicable","source_ids":["evidence-official-23e72f7e93a0a633ab2b"],"source_locator":"README.md: introduction, feature list, license and Citations"},{"label":"Known versions","value":"AutoDock Vina; README cites the 1.2.0 feature expansion separately from the original 2010 method.","status":"source_checked","source_ids":["evidence-official-23e72f7e93a0a633ab2b"],"source_locator":"README.md: introduction, feature list, license and Citations"},{"label":"Training data","value":"Not a pretrained neural model. The selected scoring function and parametrization define the procedural baseline.","status":"source_checked","source_ids":["evidence-official-23e72f7e93a0a633ab2b"],"source_locator":"README.md: introduction, feature list, license and Citations"},{"label":"Training cutoff","value":"Inapplicable to neural pretraining; reference-database and input-data dates must be recorded for each run.","status":"inapplicable","source_ids":["evidence-official-23e72f7e93a0a633ab2b"],"source_locator":"README.md: introduction, feature list, license and Citations"},{"label":"Context limits","value":"Molecular geometry and search-box constraints rather than a sequence context window.","status":"source_checked","source_ids":["evidence-official-23e72f7e93a0a633ab2b"],"source_locator":"README.md: introduction, feature list, license and Citations"},{"label":"Weights licence","value":"Inapplicable: no neural model-weight checkpoint.","status":"inapplicable","source_ids":["evidence-official-23e72f7e93a0a633ab2b"],"source_locator":"README.md: introduction, feature list, license and Citations"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/ccsb-scripps/AutoDock-Vina","status":"source_checked","source_ids":["evidence-official-23e72f7e93a0a633ab2b"],"source_locator":"README.md: introduction, feature list, license and Citations"},{"label":"Code licence","value":"Apache-2.0","status":"source_checked","source_ids":["evidence-official-640665f30e62eed9319b"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"Supports Vina and AutoDock4 scoring, multiple-ligand/batch workflows, macrocycles and Python bindings.","source_ids":["evidence-official-23e72f7e93a0a633ab2b"],"source_locator":"README.md: introduction, feature list, license and Citations"}],"limitations":[{"text":"The chosen scoring function, molecular preparation and search configuration form part of the method. A docking score is a computational quantity rather than an experimental affinity measurement.","source_ids":["evidence-official-23e72f7e93a0a633ab2b"],"source_locator":"README.md: introduction, feature list, license and Citations"}],"diagram":{"title":"AutoDock Vina workflow","steps":["Prepared receptor and ligand","Conformational search","Docking score","Ranked candidate poses"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-23e72f7e93a0a633ab2b"],"source_locator":"README.md: introduction, feature list, license and Citations"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}},"description":"Procedural docking and virtual screening","facets":{"areas":["molecular-interactions"]},"id":"discovery-model-autodock-vina","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-posebusters"}],"name":"AutoDock Vina","source_ids":["src-discovery-ccsb-scripps-autodock-vina"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":[],"entity_level":"family","reported_name":"Basenji","version":null,"historical_missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"Basenji predicts quantitative regulatory activity along DNA and scores the effects of sequence changes.","summary_source_ids":["evidence-official-dc53e8df56bb7352c014","evidence-official-edce0c06db05117706cb"],"summary_source_locator":"README.md: Basenji and Basset successor","sections":[{"title":"How it works","body":"Basenji predicts quantitative regulatory activity along DNA and scores the effects of sequence changes. Deep convolutional sequence model with binned regression outputs. The documented inputs are DNA sequence and, for training, aligned quantitative regulatory measurements. The output consists of predicted regulatory signal across sequence bins and derived variant-effect scores.","source_ids":["evidence-official-dc53e8df56bb7352c014","evidence-official-edce0c06db05117706cb"],"source_locator":"README.md: Basenji and Basset successor"},{"title":"Versions and reproducibility","body":"Basenji implementation; Basset, Akita and Saluki are described separately. The applicable input limits require configuration-specific checking.","source_ids":["evidence-official-dc53e8df56bb7352c014","evidence-official-edce0c06db05117706cb"],"source_locator":"README.md: Basenji and Basset successor"}],"facts":[{"label":"Model type","value":"Dilated convolutional genomic-track predictor","status":"source_checked","source_ids":["evidence-official-dc53e8df56bb7352c014","evidence-official-edce0c06db05117706cb"],"source_locator":"README.md: Basenji and Basset successor"},{"label":"Architecture","value":"Deep convolutional sequence model with binned regression outputs.","status":"source_checked","source_ids":["evidence-official-dc53e8df56bb7352c014","evidence-official-edce0c06db05117706cb"],"source_locator":"README.md: Basenji and Basset successor"},{"label":"Inputs","value":"DNA sequence and, for training, aligned quantitative regulatory measurements.","status":"source_checked","source_ids":["evidence-official-dc53e8df56bb7352c014","evidence-official-edce0c06db05117706cb"],"source_locator":"README.md: Basenji and Basset successor"},{"label":"Outputs","value":"Predicted regulatory signal across sequence bins and derived variant-effect scores.","status":"source_checked","source_ids":["evidence-official-dc53e8df56bb7352c014","evidence-official-edce0c06db05117706cb"],"source_locator":"README.md: Basenji and Basset successor"},{"label":"Parameters","value":"The repository describes a configurable Basenji model family rather than one checkpoint with a common parameter count; the selected model configuration is required.","status":"unreported","source_ids":["evidence-official-dc53e8df56bb7352c014","evidence-official-edce0c06db05117706cb"],"source_locator":"README.md: Basenji and Basset successor"},{"label":"Known versions","value":"Basenji implementation; Basset, Akita and Saluki are described separately.","status":"source_checked","source_ids":["evidence-official-dc53e8df56bb7352c014","evidence-official-edce0c06db05117706cb"],"source_locator":"README.md: Basenji and Basset successor"},{"label":"Training data","value":"Chosen regulatory-activity datasets; the README points to preprocessing/training tutorials rather than identifying one universal checkpoint.","status":"source_checked","source_ids":["evidence-official-dc53e8df56bb7352c014","evidence-official-edce0c06db05117706cb"],"source_locator":"README.md: Basenji and Basset successor"},{"label":"Training cutoff","value":"No checkpoint is selected by this family record. Training-track accessions and collection dates must be taken from the chosen Basenji release, not inferred from the repository date.","status":"unreported","source_ids":["evidence-official-dc53e8df56bb7352c014","evidence-official-edce0c06db05117706cb"],"source_locator":"README.md: Basenji and Basset successor"},{"label":"Context limits","value":"The inspected family README does not fix one input and output window across Basenji configurations. A run must preserve its sequence length, pooling and output-bin settings.","status":"unreported","source_ids":["evidence-official-dc53e8df56bb7352c014","evidence-official-edce0c06db05117706cb"],"source_locator":"README.md: Basenji and Basset successor"},{"label":"Weights licence","value":"Separate checkpoint-distribution terms are not stated in the inspected release documentation and licence material. The source-code licence alone is not recorded as an explicit weight grant.","status":"unreported","source_ids":["evidence-official-dc53e8df56bb7352c014","evidence-official-edce0c06db05117706cb","evidence-official-fe7d7d8a007d344f589c"],"source_locator":"README.md: Basenji and Basset successor; LICENSE: licence text"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/calico/basenji","status":"source_checked","source_ids":["evidence-official-dc53e8df56bb7352c014","evidence-official-edce0c06db05117706cb"],"source_locator":"README.md: Basenji and Basset successor"},{"label":"Code licence","value":"Apache-2.0","status":"source_checked","source_ids":["evidence-official-fe7d7d8a007d344f589c"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"The same framework supports quantitative signal prediction and nucleotide/variant attribution workflows.","source_ids":["evidence-official-dc53e8df56bb7352c014","evidence-official-edce0c06db05117706cb"],"source_locator":"README.md: Basenji and Basset successor"}],"limitations":[{"text":"The repository also contains Akita and Saluki, which are separate models. Their inputs and tasks must not be assigned to Basenji simply because they share a repository.","source_ids":["evidence-official-dc53e8df56bb7352c014","evidence-official-edce0c06db05117706cb"],"source_locator":"README.md: Basenji and Basset successor"}],"diagram":{"title":"Basenji workflow","steps":["DNA sequence","Convolutional sequence model","Binned regulatory signal","Optional variant comparison"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-dc53e8df56bb7352c014","evidence-official-edce0c06db05117706cb"],"source_locator":"README.md: Basenji and Basset successor"},"coverage":"limited","gaps":["Parameters: The repository describes a configurable Basenji model family rather than one checkpoint with a common parameter count; the selected model configuration is required.","Training cutoff: No checkpoint is selected by this family record. Training-track accessions and collection dates must be taken from the chosen Basenji release, not inferred from the repository date.","Context limits: The inspected family README does not fix one input and output window across Basenji configurations. A run must preserve its sequence length, pooling and output-bin settings.","Weights licence: Separate checkpoint-distribution terms are not stated in the inspected release documentation and licence material. The source-code licence alone is not recorded as an explicit weight grant."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}},"description":"Sequence-to-regulatory-profile prediction","facets":{"areas":["genomics"]},"id":"discovery-model-basenji","kind":"model","links":[],"name":"Basenji","source_ids":["src-discovery-calico-basenji"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-plinder"],"entity_level":"family","reported_name":"Boltz","version":null,"historical_missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"Boltz predicts biomolecular complex structures; Boltz-2 also predicts binding affinity.","summary_source_ids":["evidence-official-f780521ccebbb0691f9e","evidence-official-5a9e55abdf288d9dd8da","evidence-official-39a15ed8072aeea14a22","evidence-official-734117ed8501eb26fefc","evidence-official-7cb8cb7f091536b92c11","evidence-official-bd6ec1f819e22f075c32"],"summary_source_locator":"src/boltz/model/models/boltz2.py: Boltz2.__init__, forward, PairformerModule, AtomDiffusion and AffinityModule; docs/prediction.md: affinity outputs; Boltz-2 report Sections 2 Data, 3 Architecture, 4 Training and 6 Limitations","sections":[{"title":"How it works","body":"Boltz-2 first encodes the molecular inputs, alignments and optional templates into token and pair features. A Pairformer trunk updates those features and conditions atom-coordinate diffusion to generate a complex. Separate confidence and affinity modules assess the prediction; affinity classification and regression outputs answer different questions. This describes the Boltz-2 generation; a Boltz-1 result must retain its original checkpoint and prediction procedure.","source_ids":["evidence-official-f780521ccebbb0691f9e","evidence-official-5a9e55abdf288d9dd8da","evidence-official-39a15ed8072aeea14a22","evidence-official-734117ed8501eb26fefc","evidence-official-7cb8cb7f091536b92c11","evidence-official-bd6ec1f819e22f075c32"],"source_locator":"src/boltz/model/models/boltz2.py: Boltz2.__init__, forward, PairformerModule, AtomDiffusion and AffinityModule; docs/prediction.md: affinity outputs; Boltz-2 report Sections 2 Data, 3 Architecture, 4 Training and 6 Limitations"},{"title":"Versions and reproducibility","body":"Boltz-1 and Boltz-2 are distinct released generations; the catalogue does not select an evaluated checkpoint. The Boltz-2 report describes training crops up to 768 tokens. This is a training-crop size rather than a universal inference maximum; affinity additionally uses a pocket crop.","source_ids":["evidence-official-f780521ccebbb0691f9e","evidence-official-5a9e55abdf288d9dd8da","evidence-official-39a15ed8072aeea14a22","evidence-official-734117ed8501eb26fefc","evidence-official-7cb8cb7f091536b92c11","evidence-official-bd6ec1f819e22f075c32"],"source_locator":"src/boltz/model/models/boltz2.py: Boltz2.__init__, forward, PairformerModule, AtomDiffusion and AffinityModule; docs/prediction.md: affinity outputs; Boltz-2 report Sections 2 Data, 3 Architecture, 4 Training and 6 Limitations"}],"facts":[{"label":"Model type","value":"Biomolecular structure predictor family","status":"source_checked","source_ids":["evidence-official-f780521ccebbb0691f9e","evidence-official-5a9e55abdf288d9dd8da","evidence-official-39a15ed8072aeea14a22","evidence-official-734117ed8501eb26fefc","evidence-official-7cb8cb7f091536b92c11","evidence-official-bd6ec1f819e22f075c32"],"source_locator":"src/boltz/model/models/boltz2.py: Boltz2.__init__, forward, PairformerModule, AtomDiffusion and AffinityModule; docs/prediction.md: affinity outputs; Boltz-2 report Sections 2 Data, 3 Architecture, 4 Training and 6 Limitations"},{"label":"Architecture","value":"Boltz-1 and Boltz-2 are separate generations. In the inspected Boltz-2 implementation, molecular/MSA/template embeddings enter a Pairformer trunk, which conditions atom-coordinate diffusion; confidence and affinity are separate output modules.","status":"source_checked","source_ids":["evidence-official-f780521ccebbb0691f9e","evidence-official-5a9e55abdf288d9dd8da","evidence-official-39a15ed8072aeea14a22","evidence-official-734117ed8501eb26fefc","evidence-official-7cb8cb7f091536b92c11","evidence-official-bd6ec1f819e22f075c32"],"source_locator":"src/boltz/model/models/boltz2.py: Boltz2.__init__, forward, PairformerModule, AtomDiffusion and AffinityModule; docs/prediction.md: affinity outputs; Boltz-2 report Sections 2 Data, 3 Architecture, 4 Training and 6 Limitations"},{"label":"Inputs","value":"Protein, nucleic-acid and ligand specifications in prediction input files.","status":"source_checked","source_ids":["evidence-official-f780521ccebbb0691f9e","evidence-official-5a9e55abdf288d9dd8da","evidence-official-39a15ed8072aeea14a22","evidence-official-734117ed8501eb26fefc","evidence-official-7cb8cb7f091536b92c11","evidence-official-bd6ec1f819e22f075c32"],"source_locator":"src/boltz/model/models/boltz2.py: Boltz2.__init__, forward, PairformerModule, AtomDiffusion and AffinityModule; docs/prediction.md: affinity outputs; Boltz-2 report Sections 2 Data, 3 Architecture, 4 Training and 6 Limitations"},{"label":"Outputs","value":"Predicted complex structures and, for supported Boltz-2 inputs, binding-affinity predictions.","status":"source_checked","source_ids":["evidence-official-f780521ccebbb0691f9e","evidence-official-5a9e55abdf288d9dd8da","evidence-official-39a15ed8072aeea14a22","evidence-official-734117ed8501eb26fefc","evidence-official-7cb8cb7f091536b92c11","evidence-official-bd6ec1f819e22f075c32"],"source_locator":"src/boltz/model/models/boltz2.py: Boltz2.__init__, forward, PairformerModule, AtomDiffusion and AffinityModule; docs/prediction.md: affinity outputs; Boltz-2 report Sections 2 Data, 3 Architecture, 4 Training and 6 Limitations"},{"label":"Parameters","value":"A complete parameter total is not stated in the reviewed Boltz-2 architecture report or model constructor; structure, confidence and affinity are separate modules.","status":"unreported","source_ids":["evidence-official-f780521ccebbb0691f9e","evidence-official-5a9e55abdf288d9dd8da","evidence-official-39a15ed8072aeea14a22","evidence-official-734117ed8501eb26fefc","evidence-official-7cb8cb7f091536b92c11","evidence-official-bd6ec1f819e22f075c32"],"source_locator":"src/boltz/model/models/boltz2.py: Boltz2.__init__, forward, PairformerModule, AtomDiffusion and AffinityModule; docs/prediction.md: affinity outputs; Boltz-2 report Sections 2 Data, 3 Architecture, 4 Training and 6 Limitations"},{"label":"Known versions","value":"Boltz-1 and Boltz-2 are distinct released generations; the catalogue does not select an evaluated checkpoint.","status":"source_checked","source_ids":["evidence-official-f780521ccebbb0691f9e","evidence-official-5a9e55abdf288d9dd8da","evidence-official-39a15ed8072aeea14a22","evidence-official-734117ed8501eb26fefc","evidence-official-7cb8cb7f091536b92c11","evidence-official-bd6ec1f819e22f075c32"],"source_locator":"src/boltz/model/models/boltz2.py: Boltz2.__init__, forward, PairformerModule, AtomDiffusion and AffinityModule; docs/prediction.md: affinity outputs; Boltz-2 report Sections 2 Data, 3 Architecture, 4 Training and 6 Limitations"},{"label":"Training data","value":"Boltz-2 structure training combines pre-June-2023 PDB entries, MISATO/ATLAS/mdCATH molecular dynamics, and AlphaFold2/Boltz-1 distillation. Separate affinity training uses curated PubChem, ChEMBL, BindingDB, HTS, CeMM and MIDAS evidence with different regression/classification labels.","status":"source_checked","source_ids":["evidence-official-f780521ccebbb0691f9e","evidence-official-5a9e55abdf288d9dd8da","evidence-official-39a15ed8072aeea14a22","evidence-official-734117ed8501eb26fefc","evidence-official-7cb8cb7f091536b92c11","evidence-official-bd6ec1f819e22f075c32"],"source_locator":"src/boltz/model/models/boltz2.py: Boltz2.__init__, forward, PairformerModule, AtomDiffusion and AffinityModule; docs/prediction.md: affinity outputs; Boltz-2 report Sections 2 Data, 3 Architecture, 4 Training and 6 Limitations"},{"label":"Training cutoff","value":"Boltz-2 experimental PDB structures were released before 2023-06-01. This is not a shared cutoff for every affinity, MD or distilled resource, nor a Boltz-1 training cutoff.","status":"source_checked","source_ids":["evidence-official-f780521ccebbb0691f9e","evidence-official-5a9e55abdf288d9dd8da","evidence-official-39a15ed8072aeea14a22","evidence-official-734117ed8501eb26fefc","evidence-official-7cb8cb7f091536b92c11","evidence-official-bd6ec1f819e22f075c32"],"source_locator":"src/boltz/model/models/boltz2.py: Boltz2.__init__, forward, PairformerModule, AtomDiffusion and AffinityModule; docs/prediction.md: affinity outputs; Boltz-2 report Sections 2 Data, 3 Architecture, 4 Training and 6 Limitations"},{"label":"Context limits","value":"The Boltz-2 report describes training crops up to 768 tokens. This is a training-crop size rather than a universal inference maximum; affinity additionally uses a pocket crop.","status":"source_checked","source_ids":["evidence-official-f780521ccebbb0691f9e","evidence-official-5a9e55abdf288d9dd8da","evidence-official-39a15ed8072aeea14a22","evidence-official-734117ed8501eb26fefc","evidence-official-7cb8cb7f091536b92c11","evidence-official-bd6ec1f819e22f075c32"],"source_locator":"src/boltz/model/models/boltz2.py: Boltz2.__init__, forward, PairformerModule, AtomDiffusion and AffinityModule; docs/prediction.md: affinity outputs; Boltz-2 report Sections 2 Data, 3 Architecture, 4 Training and 6 Limitations"},{"label":"Weights licence","value":"MIT; the README explicitly applies this licence to code and model weights.","status":"source_checked","source_ids":["evidence-official-f780521ccebbb0691f9e","evidence-official-5a9e55abdf288d9dd8da","evidence-official-39a15ed8072aeea14a22","evidence-official-734117ed8501eb26fefc","evidence-official-7cb8cb7f091536b92c11","evidence-official-bd6ec1f819e22f075c32"],"source_locator":"src/boltz/model/models/boltz2.py: Boltz2.__init__, forward, PairformerModule, AtomDiffusion and AffinityModule; docs/prediction.md: affinity outputs; Boltz-2 report Sections 2 Data, 3 Architecture, 4 Training and 6 Limitations"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/jwohlwend/boltz","status":"source_checked","source_ids":["evidence-official-f780521ccebbb0691f9e","evidence-official-5a9e55abdf288d9dd8da","evidence-official-39a15ed8072aeea14a22","evidence-official-734117ed8501eb26fefc","evidence-official-7cb8cb7f091536b92c11","evidence-official-bd6ec1f819e22f075c32"],"source_locator":"src/boltz/model/models/boltz2.py: Boltz2.__init__, forward, PairformerModule, AtomDiffusion and AffinityModule; docs/prediction.md: affinity outputs; Boltz-2 report Sections 2 Data, 3 Architecture, 4 Training and 6 Limitations"},{"label":"Code licence","value":"MIT","status":"source_checked","source_ids":["evidence-official-8a985eabfd054f0dec4d"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"The project distributes prediction code, model weights and training instructions.","source_ids":["evidence-official-f780521ccebbb0691f9e","evidence-official-5a9e55abdf288d9dd8da","evidence-official-39a15ed8072aeea14a22","evidence-official-734117ed8501eb26fefc","evidence-official-7cb8cb7f091536b92c11","evidence-official-bd6ec1f819e22f075c32"],"source_locator":"src/boltz/model/models/boltz2.py: Boltz2.__init__, forward, PairformerModule, AtomDiffusion and AffinityModule; docs/prediction.md: affinity outputs; Boltz-2 report Sections 2 Data, 3 Architecture, 4 Training and 6 Limitations"}],"limitations":[{"text":"Affinity predictions depend on a plausible binding pose and mix biochemical endpoint types. The report notes limited handling of cofactors, water and multimeric binding partners, and substantial variation between assays.","source_ids":["evidence-official-f780521ccebbb0691f9e","evidence-official-5a9e55abdf288d9dd8da","evidence-official-39a15ed8072aeea14a22","evidence-official-734117ed8501eb26fefc","evidence-official-7cb8cb7f091536b92c11","evidence-official-bd6ec1f819e22f075c32"],"source_locator":"src/boltz/model/models/boltz2.py: Boltz2.__init__, forward, PairformerModule, AtomDiffusion and AffinityModule; docs/prediction.md: affinity outputs; Boltz-2 report Sections 2 Data, 3 Architecture, 4 Training and 6 Limitations"}],"diagram":{"title":"Boltz workflow","steps":["Molecular inputs, MSA and templates","Token and pair embeddings","Pairformer trunk","Atom-coordinate diffusion","Structure, confidence and affinity"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-f780521ccebbb0691f9e","evidence-official-5a9e55abdf288d9dd8da","evidence-official-39a15ed8072aeea14a22","evidence-official-734117ed8501eb26fefc","evidence-official-7cb8cb7f091536b92c11","evidence-official-bd6ec1f819e22f075c32"],"source_locator":"src/boltz/model/models/boltz2.py: Boltz2.__init__, forward, PairformerModule, AtomDiffusion and AffinityModule; docs/prediction.md: affinity outputs; Boltz-2 report Sections 2 Data, 3 Architecture, 4 Training and 6 Limitations"},"coverage":"limited","gaps":["Parameters: A complete parameter total is not stated in the reviewed Boltz-2 architecture report or model constructor; structure, confidence and affinity are separate modules."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}},"description":"Biomolecular structure and affinity model family","facets":{"areas":["molecular-interactions"]},"id":"discovery-model-boltz","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-plinder"}],"name":"Boltz","source_ids":["src-discovery-jwohlwend-boltz"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-cami"],"entity_level":"method","reported_name":"CAMISIM","version":null,"historical_missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"CAMISIM creates simulated microbial communities and corresponding shotgun metagenomic datasets.","summary_source_ids":["evidence-official-e604b144dfbd97bf8917"],"summary_source_locator":"README.md: overview and CAMISIM 2.0","sections":[{"title":"How it works","body":"CAMISIM creates simulated microbial communities and corresponding shotgun metagenomic datasets. Community-abundance simulation followed by metagenomic read generation; CAMISIM 2 uses a Nextflow workflow. The documented inputs are chosen genomes, community-abundance settings and simulation configuration. The output consists of simulated shotgun metagenomic datasets with known generating communities.","source_ids":["evidence-official-e604b144dfbd97bf8917"],"source_locator":"README.md: overview and CAMISIM 2.0"},{"title":"Versions and reproducibility","body":"CAMISIM 2.0 Nextflow workflow; legacy Python version retained as 1.31-final. Simulation size and read-generation configuration, not a neural token window.","source_ids":["evidence-official-e604b144dfbd97bf8917"],"source_locator":"README.md: overview and CAMISIM 2.0"}],"facts":[{"label":"Model type","value":"Microbial community simulation software","status":"source_checked","source_ids":["evidence-official-e604b144dfbd97bf8917"],"source_locator":"README.md: overview and CAMISIM 2.0"},{"label":"Architecture","value":"Community-abundance simulation followed by metagenomic read generation; CAMISIM 2 uses a Nextflow workflow.","status":"source_checked","source_ids":["evidence-official-e604b144dfbd97bf8917"],"source_locator":"README.md: overview and CAMISIM 2.0"},{"label":"Inputs","value":"Chosen genomes, community-abundance settings and simulation configuration.","status":"source_checked","source_ids":["evidence-official-e604b144dfbd97bf8917"],"source_locator":"README.md: overview and CAMISIM 2.0"},{"label":"Outputs","value":"Simulated shotgun metagenomic datasets with known generating communities.","status":"source_checked","source_ids":["evidence-official-e604b144dfbd97bf8917"],"source_locator":"README.md: overview and CAMISIM 2.0"},{"label":"Parameters","value":"Inapplicable as a neural parameter count; simulation settings must be pinned.","status":"inapplicable","source_ids":["evidence-official-e604b144dfbd97bf8917"],"source_locator":"README.md: overview and CAMISIM 2.0"},{"label":"Known versions","value":"CAMISIM 2.0 Nextflow workflow; legacy Python version retained as 1.31-final.","status":"source_checked","source_ids":["evidence-official-e604b144dfbd97bf8917"],"source_locator":"README.md: overview and CAMISIM 2.0"},{"label":"Training data","value":"No neural pretraining; selected input genomes determine the simulation source material.","status":"source_checked","source_ids":["evidence-official-e604b144dfbd97bf8917"],"source_locator":"README.md: overview and CAMISIM 2.0"},{"label":"Training cutoff","value":"Inapplicable to neural pretraining; reference-database and input-data dates must be recorded for each run.","status":"inapplicable","source_ids":["evidence-official-e604b144dfbd97bf8917"],"source_locator":"README.md: overview and CAMISIM 2.0"},{"label":"Context limits","value":"Simulation size and read-generation configuration, not a neural token window.","status":"source_checked","source_ids":["evidence-official-e604b144dfbd97bf8917"],"source_locator":"README.md: overview and CAMISIM 2.0"},{"label":"Weights licence","value":"Inapplicable: simulator rather than a pretrained predictor.","status":"inapplicable","source_ids":["evidence-official-e604b144dfbd97bf8917"],"source_locator":"README.md: overview and CAMISIM 2.0"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/CAMI-challenge/CAMISIM","status":"source_checked","source_ids":["evidence-official-e604b144dfbd97bf8917"],"source_locator":"README.md: overview and CAMISIM 2.0"},{"label":"Code licence","value":"Apache-2.0","status":"source_checked","source_ids":["evidence-official-1005311cb2f598fe63de"],"source_locator":"LICENSE.txt: licence text"}],"strengths":[{"text":"Provides controlled synthetic data useful for evaluating metagenomic analysis methods.","source_ids":["evidence-official-e604b144dfbd97bf8917"],"source_locator":"README.md: overview and CAMISIM 2.0"}],"limitations":[{"text":"Simulation realism depends on its inputs and configuration. The authors advise checking converted CAMISIM 1 configurations rather than assuming automatic compatibility.","source_ids":["evidence-official-e604b144dfbd97bf8917"],"source_locator":"README.md: overview and CAMISIM 2.0"}],"diagram":{"title":"CAMISIM workflow","steps":["Reference genomes","Community abundance model","Read simulation","Synthetic metagenome"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-e604b144dfbd97bf8917"],"source_locator":"README.md: overview and CAMISIM 2.0"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}},"description":"Microbial community and metagenome simulation","facets":{"areas":["microbiome"]},"id":"discovery-model-camisim","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-cami"}],"name":"CAMISIM","source_ids":["src-discovery-cami-challenge-camisim"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-dart-eval"],"entity_level":"family","reported_name":"ChromBPNet","version":null,"historical_missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"ChromBPNet predicts base-resolution chromatin accessibility while modeling assay-specific enzyme bias separately.","summary_source_ids":["evidence-official-a766fb9176a491dc1ec3","evidence-official-bc3f1b0fa5b0999bbfa0"],"summary_source_locator":"README.md: introductory explanation and Bias-factorized ChromBPNet training; chrombpnet/training/models/chrombpnet_with_bias_model.py: bpnet_model and model_params","sections":[{"title":"How it works","body":"ChromBPNet predicts base-resolution chromatin accessibility while modeling assay-specific enzyme bias separately. Residual dilated convolutional network with a frozen bias model and a transcription-factor sequence component. The documented inputs are DNA sequences and ATAC-seq or DNase-seq profiles for the configured assay. The output consists of predicted accessibility profiles and quantities used to study sequence contributions and variants.","source_ids":["evidence-official-a766fb9176a491dc1ec3","evidence-official-bc3f1b0fa5b0999bbfa0"],"source_locator":"README.md: introductory explanation and Bias-factorized ChromBPNet training; chrombpnet/training/models/chrombpnet_with_bias_model.py: bpnet_model and model_params"},{"title":"Versions and reproducibility","body":"Version-dependent trained models; README highlights a motif-discovery note for versions at or below 0.1.3. inputlen and outputlen are explicit model configuration fields. Record both the input sequence window and the central output window for the selected trained model.","source_ids":["evidence-official-a766fb9176a491dc1ec3","evidence-official-bc3f1b0fa5b0999bbfa0"],"source_locator":"README.md: introductory explanation and Bias-factorized ChromBPNet training; chrombpnet/training/models/chrombpnet_with_bias_model.py: bpnet_model and model_params"}],"facts":[{"label":"Model type","value":"Bias-factorized convolutional chromatin-profile predictor","status":"source_checked","source_ids":["evidence-official-a766fb9176a491dc1ec3","evidence-official-bc3f1b0fa5b0999bbfa0"],"source_locator":"README.md: introductory explanation and Bias-factorized ChromBPNet training; chrombpnet/training/models/chrombpnet_with_bias_model.py: bpnet_model and model_params"},{"label":"Architecture","value":"Residual dilated convolutional network with a frozen bias model and a transcription-factor sequence component.","status":"source_checked","source_ids":["evidence-official-a766fb9176a491dc1ec3","evidence-official-bc3f1b0fa5b0999bbfa0"],"source_locator":"README.md: introductory explanation and Bias-factorized ChromBPNet training; chrombpnet/training/models/chrombpnet_with_bias_model.py: bpnet_model and model_params"},{"label":"Inputs","value":"DNA sequences and ATAC-seq or DNase-seq profiles for the configured assay.","status":"source_checked","source_ids":["evidence-official-a766fb9176a491dc1ec3","evidence-official-bc3f1b0fa5b0999bbfa0"],"source_locator":"README.md: introductory explanation and Bias-factorized ChromBPNet training; chrombpnet/training/models/chrombpnet_with_bias_model.py: bpnet_model and model_params"},{"label":"Outputs","value":"Predicted accessibility profiles and quantities used to study sequence contributions and variants.","status":"source_checked","source_ids":["evidence-official-a766fb9176a491dc1ec3","evidence-official-bc3f1b0fa5b0999bbfa0"],"source_locator":"README.md: introductory explanation and Bias-factorized ChromBPNet training; chrombpnet/training/models/chrombpnet_with_bias_model.py: bpnet_model and model_params"},{"label":"Parameters","value":"Configuration-dependent: convolutional filter count and number of dilated layers are supplied in model_params; the bias component is separate.","status":"source_checked","source_ids":["evidence-official-a766fb9176a491dc1ec3","evidence-official-bc3f1b0fa5b0999bbfa0"],"source_locator":"README.md: introductory explanation and Bias-factorized ChromBPNet training; chrombpnet/training/models/chrombpnet_with_bias_model.py: bpnet_model and model_params"},{"label":"Known versions","value":"Version-dependent trained models; README highlights a motif-discovery note for versions at or below 0.1.3.","status":"source_checked","source_ids":["evidence-official-a766fb9176a491dc1ec3","evidence-official-bc3f1b0fa5b0999bbfa0"],"source_locator":"README.md: introductory explanation and Bias-factorized ChromBPNet training; chrombpnet/training/models/chrombpnet_with_bias_model.py: bpnet_model and model_params"},{"label":"Training data","value":"Two-stage fitting: background regions for enzyme bias, then accessibility-profile training for the sequence model.","status":"source_checked","source_ids":["evidence-official-a766fb9176a491dc1ec3","evidence-official-bc3f1b0fa5b0999bbfa0"],"source_locator":"README.md: introductory explanation and Bias-factorized ChromBPNet training; chrombpnet/training/models/chrombpnet_with_bias_model.py: bpnet_model and model_params"},{"label":"Training cutoff","value":"Inapplicable as a universal pretraining date: the workflow fits the supplied chromatin tracks and bias model; their accessions, dates and split belong to the individual evaluation.","status":"inapplicable","source_ids":["evidence-official-a766fb9176a491dc1ec3","evidence-official-bc3f1b0fa5b0999bbfa0"],"source_locator":"README.md: introductory explanation and Bias-factorized ChromBPNet training; chrombpnet/training/models/chrombpnet_with_bias_model.py: bpnet_model and model_params"},{"label":"Context limits","value":"inputlen and outputlen are explicit model configuration fields. Record both the input sequence window and the central output window for the selected trained model.","status":"source_checked","source_ids":["evidence-official-a766fb9176a491dc1ec3","evidence-official-bc3f1b0fa5b0999bbfa0"],"source_locator":"README.md: introductory explanation and Bias-factorized ChromBPNet training; chrombpnet/training/models/chrombpnet_with_bias_model.py: bpnet_model and model_params"},{"label":"Weights licence","value":"Separate checkpoint-distribution terms are not stated in the inspected release documentation and licence material. The source-code licence alone is not recorded as an explicit weight grant.","status":"unreported","source_ids":["evidence-official-a766fb9176a491dc1ec3","evidence-official-bc3f1b0fa5b0999bbfa0","evidence-official-45f05eebcf7a03ac99cf"],"source_locator":"README.md: introductory explanation and Bias-factorized ChromBPNet training; chrombpnet/training/models/chrombpnet_with_bias_model.py: bpnet_model and model_params; LICENSE: licence text"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/kundajelab/chrombpnet","status":"source_checked","source_ids":["evidence-official-a766fb9176a491dc1ec3","evidence-official-bc3f1b0fa5b0999bbfa0"],"source_locator":"README.md: introductory explanation and Bias-factorized ChromBPNet training; chrombpnet/training/models/chrombpnet_with_bias_model.py: bpnet_model and model_params"},{"label":"Code licence","value":"MIT","status":"source_checked","source_ids":["evidence-official-45f05eebcf7a03ac99cf"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"The two-stage procedure explicitly separates a learned assay-bias component from the sequence component of interest.","source_ids":["evidence-official-a766fb9176a491dc1ec3","evidence-official-bc3f1b0fa5b0999bbfa0"],"source_locator":"README.md: introductory explanation and Bias-factorized ChromBPNet training; chrombpnet/training/models/chrombpnet_with_bias_model.py: bpnet_model and model_params"}],"limitations":[{"text":"The bias model must correspond to the assay. The model is fitted to a specified context, so a checkpoint cannot be assumed to apply equally across assays or cell types.","source_ids":["evidence-official-a766fb9176a491dc1ec3","evidence-official-bc3f1b0fa5b0999bbfa0"],"source_locator":"README.md: introductory explanation and Bias-factorized ChromBPNet training; chrombpnet/training/models/chrombpnet_with_bias_model.py: bpnet_model and model_params"}],"diagram":{"title":"ChromBPNet workflow","steps":["Background assay data","Learn enzyme-bias model","Fit sequence model with frozen bias","Accessibility predictions"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-a766fb9176a491dc1ec3","evidence-official-bc3f1b0fa5b0999bbfa0"],"source_locator":"README.md: introductory explanation and Bias-factorized ChromBPNet training; chrombpnet/training/models/chrombpnet_with_bias_model.py: bpnet_model and model_params"},"coverage":"limited","gaps":["Weights licence: Separate checkpoint-distribution terms are not stated in the inspected release documentation and licence material. The source-code licence alone is not recorded as an explicit weight grant."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}},"description":"Bias-aware chromatin accessibility prediction","facets":{"areas":["genomics"]},"id":"discovery-model-chrombpnet","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-dart-eval"}],"name":"ChromBPNet","source_ids":["src-discovery-kundajelab-chrombpnet"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":[],"entity_level":"method","reported_name":"COBRApy","version":null,"historical_missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"COBRApy supports constraint-based analysis of metabolic networks, including flux balance and gene-deletion analyses.","summary_source_ids":["evidence-official-eb328c1926fbe500f86b"],"summary_source_locator":"README.rst: What is COBRApy? and License","sections":[{"title":"How it works","body":"COBRApy supports constraint-based analysis of metabolic networks, including flux balance and gene-deletion analyses. Metabolic network representation coupled to constrained optimization through a selected solver. The documented inputs are A metabolic reconstruction, reaction constraints and an analysis objective. The output consists of flux solutions, feasible flux ranges and outputs of configured perturbation/essentiality analyses.","source_ids":["evidence-official-eb328c1926fbe500f86b"],"source_locator":"README.rst: What is COBRApy? and License"},{"title":"Versions and reproducibility","body":"COBRApy software and selected metabolic reconstruction must both be versioned. Network size and solver capacity; no sequence context.","source_ids":["evidence-official-eb328c1926fbe500f86b"],"source_locator":"README.rst: What is COBRApy? and License"}],"facts":[{"label":"Model type","value":"Constraint-based metabolic modelling software","status":"source_checked","source_ids":["evidence-official-eb328c1926fbe500f86b"],"source_locator":"README.rst: What is COBRApy? and License"},{"label":"Architecture","value":"Metabolic network representation coupled to constrained optimization through a selected solver.","status":"source_checked","source_ids":["evidence-official-eb328c1926fbe500f86b"],"source_locator":"README.rst: What is COBRApy? and License"},{"label":"Inputs","value":"A metabolic reconstruction, reaction constraints and an analysis objective.","status":"source_checked","source_ids":["evidence-official-eb328c1926fbe500f86b"],"source_locator":"README.rst: What is COBRApy? and License"},{"label":"Outputs","value":"Flux solutions, feasible flux ranges and outputs of configured perturbation/essentiality analyses.","status":"source_checked","source_ids":["evidence-official-eb328c1926fbe500f86b"],"source_locator":"README.rst: What is COBRApy? and License"},{"label":"Parameters","value":"Inapplicable as a neural parameter count; reaction bounds and model constraints are scientific inputs.","status":"inapplicable","source_ids":["evidence-official-eb328c1926fbe500f86b"],"source_locator":"README.rst: What is COBRApy? and License"},{"label":"Known versions","value":"COBRApy software and selected metabolic reconstruction must both be versioned.","status":"source_checked","source_ids":["evidence-official-eb328c1926fbe500f86b"],"source_locator":"README.rst: What is COBRApy? and License"},{"label":"Training data","value":"No neural pretraining; the metabolic reconstruction and experimental constraints are supplied separately.","status":"source_checked","source_ids":["evidence-official-eb328c1926fbe500f86b"],"source_locator":"README.rst: What is COBRApy? and License"},{"label":"Training cutoff","value":"Inapplicable to neural pretraining; reference-database and input-data dates must be recorded for each run.","status":"inapplicable","source_ids":["evidence-official-eb328c1926fbe500f86b"],"source_locator":"README.rst: What is COBRApy? and License"},{"label":"Context limits","value":"Network size and solver capacity; no sequence context.","status":"source_checked","source_ids":["evidence-official-eb328c1926fbe500f86b"],"source_locator":"README.rst: What is COBRApy? and License"},{"label":"Weights licence","value":"Inapplicable: no universal neural checkpoint.","status":"inapplicable","source_ids":["evidence-official-eb328c1926fbe500f86b"],"source_locator":"README.rst: What is COBRApy? and License"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/opencobra/cobrapy","status":"source_checked","source_ids":["evidence-official-eb328c1926fbe500f86b"],"source_locator":"README.rst: What is COBRApy? and License"},{"label":"Code licence","value":"GPL-2.0-or-later or LGPL-2.0-or-later, at the user’s choice, as stated by the README.","status":"source_checked","source_ids":["evidence-official-eb328c1926fbe500f86b"],"source_locator":"README.rst: License"}],"strengths":[{"text":"One software framework exposes FBA, flux variability analysis and related methods over reusable metabolic models.","source_ids":["evidence-official-eb328c1926fbe500f86b"],"source_locator":"README.rst: What is COBRApy? and License"}],"limitations":[{"text":"Results depend on the metabolic reconstruction, constraints, objective and solver; a library version alone does not define the biological model.","source_ids":["evidence-official-eb328c1926fbe500f86b"],"source_locator":"README.rst: What is COBRApy? and License"}],"diagram":{"title":"COBRApy workflow","steps":["Metabolic network","Constraints and objective","Optimization solver","Flux or perturbation analysis"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-eb328c1926fbe500f86b"],"source_locator":"README.rst: What is COBRApy? and License"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}},"description":"Constraint-based metabolic modelling","facets":{"areas":["mechanistic-biology"]},"id":"discovery-model-cobrapy","kind":"model","links":[],"name":"COBRApy","source_ids":["src-discovery-opencobra-cobrapy"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-casp"],"entity_level":"method","reported_name":"ColabFold","version":null,"historical_missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"ColabFold packages sequence-search and protein-folding workflows for notebooks and local batch prediction.","summary_source_ids":["evidence-official-efc93d867cbf62e80ebb"],"summary_source_locator":"README.md: notebook table and FAQ","sections":[{"title":"How it works","body":"ColabFold packages sequence-search and protein-folding workflows for notebooks and local batch prediction. A prediction pipeline that can combine MMseqs2 alignments with AlphaFold-family or other folding models; it is not a single neural architecture. The documented inputs are protein sequences, optionally templates and the selected alignment procedure. The output consists of predicted structures and confidence outputs from the chosen underlying folding model.","source_ids":["evidence-official-efc93d867cbf62e80ebb"],"source_locator":"README.md: notebook table and FAQ"},{"title":"Versions and reproducibility","body":"README identifies ColabFold 1.6.3 and separately lists AlphaFold2, OpenFold3, ESMFold and beta notebooks. Hardware- and predictor-dependent. The README gives approximate 16GB-GPU guidance, not a universal model context.","source_ids":["evidence-official-efc93d867cbf62e80ebb"],"source_locator":"README.md: notebook table and FAQ"}],"facts":[{"label":"Model type","value":"Sequence-search and folding pipeline","status":"source_checked","source_ids":["evidence-official-efc93d867cbf62e80ebb"],"source_locator":"README.md: notebook table and FAQ"},{"label":"Architecture","value":"A prediction pipeline that can combine MMseqs2 alignments with AlphaFold-family or other folding models; it is not a single neural architecture.","status":"source_checked","source_ids":["evidence-official-efc93d867cbf62e80ebb"],"source_locator":"README.md: notebook table and FAQ"},{"label":"Inputs","value":"Protein sequences, optionally templates and the selected alignment procedure.","status":"source_checked","source_ids":["evidence-official-efc93d867cbf62e80ebb"],"source_locator":"README.md: notebook table and FAQ"},{"label":"Outputs","value":"Predicted structures and confidence outputs from the chosen underlying folding model.","status":"source_checked","source_ids":["evidence-official-efc93d867cbf62e80ebb"],"source_locator":"README.md: notebook table and FAQ"},{"label":"Parameters","value":"Depends on the underlying folding model; no single ColabFold parameter total.","status":"source_checked","source_ids":["evidence-official-efc93d867cbf62e80ebb"],"source_locator":"README.md: notebook table and FAQ"},{"label":"Known versions","value":"README identifies ColabFold 1.6.3 and separately lists AlphaFold2, OpenFold3, ESMFold and beta notebooks.","status":"source_checked","source_ids":["evidence-official-efc93d867cbf62e80ebb"],"source_locator":"README.md: notebook table and FAQ"},{"label":"Training data","value":"ColabFold assembles existing prediction models and search databases; their provenance must be recorded separately.","status":"source_checked","source_ids":["evidence-official-efc93d867cbf62e80ebb"],"source_locator":"README.md: notebook table and FAQ"},{"label":"Training cutoff","value":"Inapplicable as one pipeline-wide date: the selected folding weights and the MSA/template databases have separate releases and cutoffs.","status":"inapplicable","source_ids":["evidence-official-efc93d867cbf62e80ebb"],"source_locator":"README.md: notebook table and FAQ"},{"label":"Context limits","value":"Hardware- and predictor-dependent. The README gives approximate 16GB-GPU guidance, not a universal model context.","status":"source_checked","source_ids":["evidence-official-efc93d867cbf62e80ebb"],"source_locator":"README.md: notebook table and FAQ"},{"label":"Weights licence","value":"Each underlying model has its own weight licence; repository code terms do not replace those licences.","status":"source_checked","source_ids":["evidence-official-efc93d867cbf62e80ebb"],"source_locator":"README.md: notebook table and FAQ"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/sokrypton/ColabFold","status":"source_checked","source_ids":["evidence-official-efc93d867cbf62e80ebb"],"source_locator":"README.md: notebook table and FAQ"},{"label":"Code licence","value":"MIT","status":"source_checked","source_ids":["evidence-official-f3fd16387ba8459ecf34"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"Notebook and batch interfaces make several folding workflows available with documented input choices.","source_ids":["evidence-official-efc93d867cbf62e80ebb"],"source_locator":"README.md: notebook table and FAQ"}],"limitations":[{"text":"The repository distinguishes supported, beta and retired notebooks. GPU memory limits and MSA-server usage policies apply; the underlying predictor and database version must be named in a benchmark.","source_ids":["evidence-official-efc93d867cbf62e80ebb"],"source_locator":"README.md: notebook table and FAQ"}],"diagram":{"title":"ColabFold workflow","steps":["Protein sequences","Selected alignment search","Selected folding model","Structure and confidence"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-efc93d867cbf62e80ebb"],"source_locator":"README.md: notebook table and FAQ"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}},"description":"Protein folding pipeline","facets":{"areas":["protein-structure"]},"id":"discovery-model-colabfold","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-casp"}],"name":"ColabFold","source_ids":["src-discovery-sokrypton-colabfold"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-gue"],"entity_level":"family","reported_name":"DNABERT-2","version":null,"historical_missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"DNABERT-2 learns DNA representations that can be adapted to genomic prediction tasks.","summary_source_ids":["evidence-official-96b5c3a31a50f7c259d6","evidence-official-25b222d11900e0e88a51","evidence-official-3aa749d3205c0a768b22"],"summary_source_locator":"DNABERT-2 paper Sections 3.2, 4.1 and 5.2 (Further Pre-Training); official model card and implementation README; official DNABERT-2-117M checkpoint LICENSE at revision 7bce263b15377fc15361f52cfab88f8b586abda0","sections":[{"title":"How it works","body":"DNABERT-2 merges recurring DNA substrings into byte-pair tokens, then processes those tokens with a masked-language-model transformer. ALiBi supplies distance-dependent attention biases, while FlashAttention changes how attention is computed. The resulting contextual embeddings need an explicit pooling rule and prediction head for a downstream task.","source_ids":["evidence-official-96b5c3a31a50f7c259d6","evidence-official-25b222d11900e0e88a51","evidence-official-3aa749d3205c0a768b22"],"source_locator":"DNABERT-2 paper Sections 3.2, 4.1 and 5.2 (Further Pre-Training); official model card and implementation README; official DNABERT-2-117M checkpoint LICENSE at revision 7bce263b15377fc15361f52cfab88f8b586abda0"},{"title":"Versions and reproducibility","body":"DNABERT-2-117M model card and official DNABERT_2 implementation. ALiBi permits inference beyond the pretraining sequence length, subject to attention/memory cost; this does not establish unlimited biological context or validated accuracy at arbitrary lengths.","source_ids":["evidence-official-96b5c3a31a50f7c259d6","evidence-official-25b222d11900e0e88a51","evidence-official-3aa749d3205c0a768b22"],"source_locator":"DNABERT-2 paper Sections 3.2, 4.1 and 5.2 (Further Pre-Training); official model card and implementation README; official DNABERT-2-117M checkpoint LICENSE at revision 7bce263b15377fc15361f52cfab88f8b586abda0"}],"facts":[{"label":"Model type","value":"Masked-token DNA transformer encoder","status":"source_checked","source_ids":["evidence-official-96b5c3a31a50f7c259d6","evidence-official-25b222d11900e0e88a51","evidence-official-3aa749d3205c0a768b22"],"source_locator":"DNABERT-2 paper Sections 3.2, 4.1 and 5.2 (Further Pre-Training); official model card and implementation README; official DNABERT-2-117M checkpoint LICENSE at revision 7bce263b15377fc15361f52cfab88f8b586abda0"},{"label":"Architecture","value":"BERT-style DNA encoder with byte-pair tokenization, ALiBi relative attention biases and FlashAttention; task heads and pooling are separately configured.","status":"source_checked","source_ids":["evidence-official-96b5c3a31a50f7c259d6","evidence-official-25b222d11900e0e88a51","evidence-official-3aa749d3205c0a768b22"],"source_locator":"DNABERT-2 paper Sections 3.2, 4.1 and 5.2 (Further Pre-Training); official model card and implementation README; official DNABERT-2-117M checkpoint LICENSE at revision 7bce263b15377fc15361f52cfab88f8b586abda0"},{"label":"Inputs","value":"DNA sequence tokenized with the supplied tokenizer.","status":"source_checked","source_ids":["evidence-official-96b5c3a31a50f7c259d6","evidence-official-25b222d11900e0e88a51","evidence-official-3aa749d3205c0a768b22"],"source_locator":"DNABERT-2 paper Sections 3.2, 4.1 and 5.2 (Further Pre-Training); official model card and implementation README; official DNABERT-2-117M checkpoint LICENSE at revision 7bce263b15377fc15361f52cfab88f8b586abda0"},{"label":"Outputs","value":"Token representations and, after a specified adaptation, task predictions.","status":"source_checked","source_ids":["evidence-official-96b5c3a31a50f7c259d6","evidence-official-25b222d11900e0e88a51","evidence-official-3aa749d3205c0a768b22"],"source_locator":"DNABERT-2 paper Sections 3.2, 4.1 and 5.2 (Further Pre-Training); official model card and implementation README; official DNABERT-2-117M checkpoint LICENSE at revision 7bce263b15377fc15361f52cfab88f8b586abda0"},{"label":"Parameters","value":"117 million for DNABERT-2-117M; family names do not establish a particular checkpoint.","status":"source_checked","source_ids":["evidence-official-96b5c3a31a50f7c259d6","evidence-official-25b222d11900e0e88a51","evidence-official-3aa749d3205c0a768b22"],"source_locator":"DNABERT-2 paper Sections 3.2, 4.1 and 5.2 (Further Pre-Training); official model card and implementation README; official DNABERT-2-117M checkpoint LICENSE at revision 7bce263b15377fc15361f52cfab88f8b586abda0"},{"label":"Known versions","value":"DNABERT-2-117M model card and official DNABERT_2 implementation.","status":"source_checked","source_ids":["evidence-official-96b5c3a31a50f7c259d6","evidence-official-25b222d11900e0e88a51","evidence-official-3aa749d3205c0a768b22"],"source_locator":"DNABERT-2 paper Sections 3.2, 4.1 and 5.2 (Further Pre-Training); official model card and implementation README; official DNABERT-2-117M checkpoint LICENSE at revision 7bce263b15377fc15361f52cfab88f8b586abda0"},{"label":"Training data","value":"The paper describes a 32.49-billion-base corpus covering 135 species in six groups, alongside a 2.75-billion-base human corpus. Further GUE-domain pretraining is a separately reported model variant.","status":"source_checked","source_ids":["evidence-official-96b5c3a31a50f7c259d6","evidence-official-25b222d11900e0e88a51","evidence-official-3aa749d3205c0a768b22"],"source_locator":"DNABERT-2 paper Sections 3.2, 4.1 and 5.2 (Further Pre-Training); official model card and implementation README; official DNABERT-2-117M checkpoint LICENSE at revision 7bce263b15377fc15361f52cfab88f8b586abda0"},{"label":"Training cutoff","value":"The paper identifies the human and multispecies genome corpora but does not state one latest-sequence deposition date in its reviewed pretraining-data sections.","status":"unreported","source_ids":["evidence-official-96b5c3a31a50f7c259d6","evidence-official-25b222d11900e0e88a51","evidence-official-3aa749d3205c0a768b22"],"source_locator":"DNABERT-2 paper Sections 3.2, 4.1 and 5.2 (Further Pre-Training); official model card and implementation README; official DNABERT-2-117M checkpoint LICENSE at revision 7bce263b15377fc15361f52cfab88f8b586abda0"},{"label":"Context limits","value":"ALiBi permits inference beyond the pretraining sequence length, subject to attention/memory cost; this does not establish unlimited biological context or validated accuracy at arbitrary lengths.","status":"source_checked","source_ids":["evidence-official-96b5c3a31a50f7c259d6","evidence-official-25b222d11900e0e88a51","evidence-official-3aa749d3205c0a768b22"],"source_locator":"DNABERT-2 paper Sections 3.2, 4.1 and 5.2 (Further Pre-Training); official model card and implementation README; official DNABERT-2-117M checkpoint LICENSE at revision 7bce263b15377fc15361f52cfab88f8b586abda0"},{"label":"Weights licence","value":"The official zhihan1996/DNABERT-2-117M checkpoint repository carries Apache-2.0 in its pinned LICENSE. This does not assign terms to a separately fitted downstream predictor.","status":"source_checked","source_ids":["evidence-official-96b5c3a31a50f7c259d6","evidence-official-25b222d11900e0e88a51","evidence-official-3aa749d3205c0a768b22"],"source_locator":"DNABERT-2 paper Sections 3.2, 4.1 and 5.2 (Further Pre-Training); official model card and implementation README; official DNABERT-2-117M checkpoint LICENSE at revision 7bce263b15377fc15361f52cfab88f8b586abda0"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/MAGICS-LAB/DNABERT_2","status":"source_checked","source_ids":["evidence-official-96b5c3a31a50f7c259d6","evidence-official-25b222d11900e0e88a51","evidence-official-3aa749d3205c0a768b22"],"source_locator":"DNABERT-2 paper Sections 3.2, 4.1 and 5.2 (Further Pre-Training); official model card and implementation README; official DNABERT-2-117M checkpoint LICENSE at revision 7bce263b15377fc15361f52cfab88f8b586abda0"},{"label":"Code licence","value":"Apache-2.0","status":"source_checked","source_ids":["evidence-official-25b222d11900e0e88a51"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"The released model supports embedding extraction and task-specific fine-tuning.","source_ids":["evidence-official-96b5c3a31a50f7c259d6","evidence-official-25b222d11900e0e88a51","evidence-official-3aa749d3205c0a768b22"],"source_locator":"DNABERT-2 paper Sections 3.2, 4.1 and 5.2 (Further Pre-Training); official model card and implementation README; official DNABERT-2-117M checkpoint LICENSE at revision 7bce263b15377fc15361f52cfab88f8b586abda0"}],"limitations":[{"text":"An embedding model alone is not the same evaluated pipeline as frozen embeddings followed by logistic regression. Tokenization and pooling choices must be preserved.","source_ids":["evidence-official-96b5c3a31a50f7c259d6","evidence-official-25b222d11900e0e88a51","evidence-official-3aa749d3205c0a768b22"],"source_locator":"DNABERT-2 paper Sections 3.2, 4.1 and 5.2 (Further Pre-Training); official model card and implementation README; official DNABERT-2-117M checkpoint LICENSE at revision 7bce263b15377fc15361f52cfab88f8b586abda0"}],"diagram":{"title":"DNABERT-2 workflow","steps":["DNA sequence","BPE tokens","Transformer encoder","Representations","Specified task head"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-96b5c3a31a50f7c259d6","evidence-official-25b222d11900e0e88a51","evidence-official-3aa749d3205c0a768b22"],"source_locator":"DNABERT-2 paper Sections 3.2, 4.1 and 5.2 (Further Pre-Training); official model card and implementation README; official DNABERT-2-117M checkpoint LICENSE at revision 7bce263b15377fc15361f52cfab88f8b586abda0"},"coverage":"limited","gaps":["Training cutoff: The paper identifies the human and multispecies genome corpora but does not state one latest-sequence deposition date in its reviewed pretraining-data sections."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}},"description":"Genomic sequence representation model","facets":{"areas":["genomics"]},"id":"discovery-model-dnabert-2","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-gue"}],"name":"DNABERT-2","source_ids":["src-discovery-magics-lab-dnabert-2"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-massspecgym"],"entity_level":"family","reported_name":"DreaMS","version":null,"historical_missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"DreaMS learns molecular representations from tandem mass spectra using self-supervised learning.","summary_source_ids":["evidence-official-89de5de6fecdda12060d","evidence-official-3076c52b2fa7cb48ba7b","evidence-official-116050b81d03a38e3b6f"],"summary_source_locator":"Paper: DreaMS neural network architecture and Hyperparameters, ablation studies, implementation details and benchmarking; README.md: models/data links; Zenodo record 10997887 metadata.license and files; DreaMS paper Introduction and GeMS mining methods","sections":[{"title":"How it works","body":"DreaMS learns molecular representations from tandem mass spectra using self-supervised learning. PeakEncoder maps spectral peaks to continuous features; SpectrumEncoder uses transformer blocks; a task-specific PeakDecoder maps the contextual features to predictions. The documented inputs are MS/MS spectra with the required peak and acquisition information. The output consists of spectrum embeddings or predictions from separately fine-tuned spectral tasks.","source_ids":["evidence-official-89de5de6fecdda12060d","evidence-official-3076c52b2fa7cb48ba7b","evidence-official-116050b81d03a38e3b6f"],"source_locator":"Paper: DreaMS neural network architecture and Hyperparameters, ablation studies, implementation details and benchmarking; README.md: models/data links; Zenodo record 10997887 metadata.license and files; DreaMS paper Introduction and GeMS mining methods"},{"title":"Versions and reproducibility","body":"Pretrained model and task-specific fine-tunes distributed separately via the linked Zenodo release. The paper describes retaining 60 spectral peaks for the transformer; this is peak-count preprocessing, not a nucleotide or protein context.","source_ids":["evidence-official-89de5de6fecdda12060d","evidence-official-3076c52b2fa7cb48ba7b","evidence-official-116050b81d03a38e3b6f"],"source_locator":"Paper: DreaMS neural network architecture and Hyperparameters, ablation studies, implementation details and benchmarking; README.md: models/data links; Zenodo record 10997887 metadata.license and files; DreaMS paper Introduction and GeMS mining methods"}],"facts":[{"label":"Model type","value":"Tandem mass-spectral transformer","status":"source_checked","source_ids":["evidence-official-89de5de6fecdda12060d","evidence-official-3076c52b2fa7cb48ba7b","evidence-official-116050b81d03a38e3b6f"],"source_locator":"Paper: DreaMS neural network architecture and Hyperparameters, ablation studies, implementation details and benchmarking; README.md: models/data links; Zenodo record 10997887 metadata.license and files; DreaMS paper Introduction and GeMS mining methods"},{"label":"Architecture","value":"PeakEncoder maps spectral peaks to continuous features; SpectrumEncoder uses transformer blocks; a task-specific PeakDecoder maps the contextual features to predictions.","status":"source_checked","source_ids":["evidence-official-89de5de6fecdda12060d","evidence-official-3076c52b2fa7cb48ba7b","evidence-official-116050b81d03a38e3b6f"],"source_locator":"Paper: DreaMS neural network architecture and Hyperparameters, ablation studies, implementation details and benchmarking; README.md: models/data links; Zenodo record 10997887 metadata.license and files; DreaMS paper Introduction and GeMS mining methods"},{"label":"Inputs","value":"MS/MS spectra with the required peak and acquisition information.","status":"source_checked","source_ids":["evidence-official-89de5de6fecdda12060d","evidence-official-3076c52b2fa7cb48ba7b","evidence-official-116050b81d03a38e3b6f"],"source_locator":"Paper: DreaMS neural network architecture and Hyperparameters, ablation studies, implementation details and benchmarking; README.md: models/data links; Zenodo record 10997887 metadata.license and files; DreaMS paper Introduction and GeMS mining methods"},{"label":"Outputs","value":"Spectrum embeddings or predictions from separately fine-tuned spectral tasks.","status":"source_checked","source_ids":["evidence-official-89de5de6fecdda12060d","evidence-official-3076c52b2fa7cb48ba7b","evidence-official-116050b81d03a38e3b6f"],"source_locator":"Paper: DreaMS neural network architecture and Hyperparameters, ablation studies, implementation details and benchmarking; README.md: models/data links; Zenodo record 10997887 metadata.license and files; DreaMS paper Introduction and GeMS mining methods"},{"label":"Parameters","value":"116M for the complete self-supervised network reported in the DreaMS paper; downstream embedding-only configurations may have fewer parameters.","status":"source_checked","source_ids":["evidence-official-89de5de6fecdda12060d","evidence-official-3076c52b2fa7cb48ba7b","evidence-official-116050b81d03a38e3b6f"],"source_locator":"Paper: DreaMS neural network architecture and Hyperparameters, ablation studies, implementation details and benchmarking; README.md: models/data links; Zenodo record 10997887 metadata.license and files; DreaMS paper Introduction and GeMS mining methods"},{"label":"Known versions","value":"Pretrained model and task-specific fine-tunes distributed separately via the linked Zenodo release.","status":"source_checked","source_ids":["evidence-official-89de5de6fecdda12060d","evidence-official-3076c52b2fa7cb48ba7b","evidence-official-116050b81d03a38e3b6f"],"source_locator":"Paper: DreaMS neural network architecture and Hyperparameters, ablation studies, implementation details and benchmarking; README.md: models/data links; Zenodo record 10997887 metadata.license and files; DreaMS paper Introduction and GeMS mining methods"},{"label":"Training data","value":"GeMS mined from MassIVE/GNPS; the paper identifies GeMS-A10, approximately 24M spectra, as the high-quality pretraining subset.","status":"source_checked","source_ids":["evidence-official-89de5de6fecdda12060d","evidence-official-3076c52b2fa7cb48ba7b","evidence-official-116050b81d03a38e3b6f"],"source_locator":"Paper: DreaMS neural network architecture and Hyperparameters, ablation studies, implementation details and benchmarking; README.md: models/data links; Zenodo record 10997887 metadata.license and files; DreaMS paper Introduction and GeMS mining methods"},{"label":"Training cutoff","value":"GeMS mining selected GNPS-tagged MassIVE studies available as of November 2022; subsequent task-specific datasets have separate provenance.","status":"source_checked","source_ids":["evidence-official-89de5de6fecdda12060d","evidence-official-3076c52b2fa7cb48ba7b","evidence-official-116050b81d03a38e3b6f"],"source_locator":"Paper: DreaMS neural network architecture and Hyperparameters, ablation studies, implementation details and benchmarking; README.md: models/data links; Zenodo record 10997887 metadata.license and files; DreaMS paper Introduction and GeMS mining methods"},{"label":"Context limits","value":"The paper describes retaining 60 spectral peaks for the transformer; this is peak-count preprocessing, not a nucleotide or protein context.","status":"source_checked","source_ids":["evidence-official-89de5de6fecdda12060d","evidence-official-3076c52b2fa7cb48ba7b","evidence-official-116050b81d03a38e3b6f"],"source_locator":"Paper: DreaMS neural network architecture and Hyperparameters, ablation studies, implementation details and benchmarking; README.md: models/data links; Zenodo record 10997887 metadata.license and files; DreaMS paper Introduction and GeMS mining methods"},{"label":"Weights licence","value":"CC-BY-4.0 for the embedding_model.ckpt and ssl_model.ckpt files in author-linked Zenodo record 10997887; separate from the MIT code licence.","status":"source_checked","source_ids":["evidence-official-89de5de6fecdda12060d","evidence-official-3076c52b2fa7cb48ba7b","evidence-official-116050b81d03a38e3b6f"],"source_locator":"Paper: DreaMS neural network architecture and Hyperparameters, ablation studies, implementation details and benchmarking; README.md: models/data links; Zenodo record 10997887 metadata.license and files; DreaMS paper Introduction and GeMS mining methods"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/pluskal-lab/DreaMS","status":"source_checked","source_ids":["evidence-official-89de5de6fecdda12060d","evidence-official-3076c52b2fa7cb48ba7b","evidence-official-116050b81d03a38e3b6f"],"source_locator":"Paper: DreaMS neural network architecture and Hyperparameters, ablation studies, implementation details and benchmarking; README.md: models/data links; Zenodo record 10997887 metadata.license and files; DreaMS paper Introduction and GeMS mining methods"},{"label":"Code licence","value":"MIT","status":"source_checked","source_ids":["evidence-official-f6c94a04c51af984a008"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"The same pretrained representation can be adapted to spectral similarity and molecular annotation tasks.","source_ids":["evidence-official-89de5de6fecdda12060d","evidence-official-3076c52b2fa7cb48ba7b","evidence-official-116050b81d03a38e3b6f"],"source_locator":"Paper: DreaMS neural network architecture and Hyperparameters, ablation studies, implementation details and benchmarking; README.md: models/data links; Zenodo record 10997887 metadata.license and files; DreaMS paper Introduction and GeMS mining methods"}],"limitations":[{"text":"Pretrained representations and fine-tuned similarity/property predictors are distinct models. Spectrum quality and preprocessing remain essential parts of an evaluation.","source_ids":["evidence-official-89de5de6fecdda12060d","evidence-official-3076c52b2fa7cb48ba7b","evidence-official-116050b81d03a38e3b6f"],"source_locator":"Paper: DreaMS neural network architecture and Hyperparameters, ablation studies, implementation details and benchmarking; README.md: models/data links; Zenodo record 10997887 metadata.license and files; DreaMS paper Introduction and GeMS mining methods"}],"diagram":{"title":"DreaMS workflow","steps":["MS/MS spectrum","Peak representation","Pretrained transformer","Embedding or fine-tuned prediction"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-89de5de6fecdda12060d","evidence-official-3076c52b2fa7cb48ba7b","evidence-official-116050b81d03a38e3b6f"],"source_locator":"Paper: DreaMS neural network architecture and Hyperparameters, ablation studies, implementation details and benchmarking; README.md: models/data links; Zenodo record 10997887 metadata.license and files; DreaMS paper Introduction and GeMS mining methods"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}},"description":"Tandem mass spectrum representation model","facets":{"areas":["metabolomics"]},"id":"discovery-model-dreams","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-massspecgym"}],"name":"DreaMS","source_ids":["src-discovery-pluskal-lab-dreams"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-proteingym"],"entity_level":"family","reported_name":"ESM-1v","version":null,"historical_missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"ESM-1v provides protein language models intended for zero-shot variant-effect scoring.","summary_source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-137a3a936ffa4673b3d3"],"summary_source_locator":"README.md: Main models, Zero-shot variant prediction and Pre-trained Models","sections":[{"title":"How it works","body":"ESM-1v provides protein language models intended for zero-shot variant-effect scoring. Transformer architecture shared with ESM-1b, trained on UniRef90; multiple released model instances support the variant-scoring workflow. The documented inputs are protein amino-acid sequence and the specified sequence variation. The output consists of token probabilities used to calculate sequence-variant scores.","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-137a3a936ffa4673b3d3"],"source_locator":"README.md: Main models, Zero-shot variant prediction and Pre-trained Models"},{"title":"Versions and reproducibility","body":"esm1v_t33_650M_UR90S_1 through _5. The applicable input limits require configuration-specific checking.","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-137a3a936ffa4673b3d3"],"source_locator":"README.md: Main models, Zero-shot variant prediction and Pre-trained Models"}],"facts":[{"label":"Model type","value":"Protein transformer for variant-effect scoring","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-137a3a936ffa4673b3d3"],"source_locator":"README.md: Main models, Zero-shot variant prediction and Pre-trained Models"},{"label":"Architecture","value":"Transformer architecture shared with ESM-1b, trained on UniRef90; multiple released model instances support the variant-scoring workflow.","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-137a3a936ffa4673b3d3"],"source_locator":"README.md: Main models, Zero-shot variant prediction and Pre-trained Models"},{"label":"Inputs","value":"Protein amino-acid sequence and the specified sequence variation.","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-137a3a936ffa4673b3d3"],"source_locator":"README.md: Main models, Zero-shot variant prediction and Pre-trained Models"},{"label":"Outputs","value":"Token probabilities used to calculate sequence-variant scores.","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-137a3a936ffa4673b3d3"],"source_locator":"README.md: Main models, Zero-shot variant prediction and Pre-trained Models"},{"label":"Parameters","value":"650 million per model; 33 transformer layers.","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-137a3a936ffa4673b3d3"],"source_locator":"README.md: Main models, Zero-shot variant prediction and Pre-trained Models"},{"label":"Known versions","value":"esm1v_t33_650M_UR90S_1 through _5.","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-137a3a936ffa4673b3d3"],"source_locator":"README.md: Main models, Zero-shot variant prediction and Pre-trained Models"},{"label":"Training data","value":"UniRef90/S 2020_03, as reported in the pretrained-model table.","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-137a3a936ffa4673b3d3"],"source_locator":"README.md: Main models, Zero-shot variant prediction and Pre-trained Models"},{"label":"Training cutoff","value":"The official model table identifies the UniRef90/S2020_03 training release; this is a corpus version rather than a verified latest-sequence deposition date.","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-137a3a936ffa4673b3d3"],"source_locator":"README.md: Main models, Zero-shot variant prediction and Pre-trained Models"},{"label":"Context limits","value":"The inspected official esm1v_t33_650M_UR90S_1 config reserves 1,026 absolute positions. This includes model positions and is not a claim of 1,026-residue validated biological context.","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-137a3a936ffa4673b3d3"],"source_locator":"README.md: Main models, Zero-shot variant prediction and Pre-trained Models"},{"label":"Weights licence","value":"Separate checkpoint-distribution terms are not stated in the inspected release documentation and licence material. The source-code licence alone is not recorded as an explicit weight grant.","status":"unreported","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-137a3a936ffa4673b3d3","evidence-official-2e7c7649620407f50f6b"],"source_locator":"README.md: Main models, Zero-shot variant prediction and Pre-trained Models; LICENSE: licence text"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/facebookresearch/esm","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-137a3a936ffa4673b3d3"],"source_locator":"README.md: Main models, Zero-shot variant prediction and Pre-trained Models"},{"label":"Code licence","value":"MIT","status":"source_checked","source_ids":["evidence-official-2e7c7649620407f50f6b"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"The package provides a dedicated variant-prediction example without fitting a task-specific supervised classifier.","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-137a3a936ffa4673b3d3"],"source_locator":"README.md: Main models, Zero-shot variant prediction and Pre-trained Models"}],"limitations":[{"text":"The five released instances are distinct models. A language-model score is not a calibrated clinical pathogenicity probability.","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-137a3a936ffa4673b3d3"],"source_locator":"README.md: Main models, Zero-shot variant prediction and Pre-trained Models"}],"diagram":{"title":"ESM-1v workflow","steps":["Reference protein sequence","ESM-1v token predictions","Specified variant scoring rule","Variant score"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-137a3a936ffa4673b3d3"],"source_locator":"README.md: Main models, Zero-shot variant prediction and Pre-trained Models"},"coverage":"limited","gaps":["Weights licence: Separate checkpoint-distribution terms are not stated in the inspected release documentation and licence material. The source-code licence alone is not recorded as an explicit weight grant."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}},"description":"Protein variant effect model family","facets":{"areas":["protein-function"]},"id":"discovery-model-esm-1v","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-proteingym"}],"name":"ESM-1v","source_ids":["src-discovery-facebookresearch-esm"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-proteingym"],"entity_level":"family","reported_name":"ESM-2","version":null,"historical_missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"ESM-2 is a family of protein sequence encoders that produce representations for downstream protein analyses.","summary_source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-d7e5c63feb0e0e627625","evidence-official-28fc17fbe0c1d99da219"],"summary_source_locator":"ESM README: Pre-trained Models and Main models; official facebook/esm2_t33_650M_UR50D README licence metadata and config.json","sections":[{"title":"How it works","body":"ESM-2 tokenizes an amino-acid sequence and uses a transformer encoder trained to recover masked residues. Self-attention lets each residue representation depend on its sequence context. The released model returns token probabilities and embeddings; a specified pooling rule, task head or complete folding pipeline is needed for a particular biological prediction.","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-d7e5c63feb0e0e627625","evidence-official-28fc17fbe0c1d99da219"],"source_locator":"ESM README: Pre-trained Models and Main models; official facebook/esm2_t33_650M_UR50D README licence metadata and config.json"},{"title":"Versions and reproducibility","body":"ESM-2 checkpoint identifiers encode layer count, parameter scale and training-data tag. The checked esm2_t33_650M_UR50D configuration lists max_position_embeddings=1,026. This configuration field includes model positions and is not a claim of training or validated inference on 1,026 amino acids.","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-d7e5c63feb0e0e627625","evidence-official-28fc17fbe0c1d99da219"],"source_locator":"ESM README: Pre-trained Models and Main models; official facebook/esm2_t33_650M_UR50D README licence metadata and config.json"}],"facts":[{"label":"Model type","value":"Masked-token protein transformer encoder","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-d7e5c63feb0e0e627625","evidence-official-28fc17fbe0c1d99da219"],"source_locator":"ESM README: Pre-trained Models and Main models; official facebook/esm2_t33_650M_UR50D README licence metadata and config.json"},{"label":"Architecture","value":"Masked-token protein transformer encoder; the checked 650M checkpoint has 33 layers, hidden width 1,280, 20 attention heads and rotary positional encoding.","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-d7e5c63feb0e0e627625","evidence-official-28fc17fbe0c1d99da219"],"source_locator":"ESM README: Pre-trained Models and Main models; official facebook/esm2_t33_650M_UR50D README licence metadata and config.json"},{"label":"Inputs","value":"Single amino-acid sequences.","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-d7e5c63feb0e0e627625","evidence-official-28fc17fbe0c1d99da219"],"source_locator":"ESM README: Pre-trained Models and Main models; official facebook/esm2_t33_650M_UR50D README licence metadata and config.json"},{"label":"Outputs","value":"Residue embeddings, sequence representations and masked-token predictions.","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-d7e5c63feb0e0e627625","evidence-official-28fc17fbe0c1d99da219"],"source_locator":"ESM README: Pre-trained Models and Main models; official facebook/esm2_t33_650M_UR50D README licence metadata and config.json"},{"label":"Parameters","value":"Released scales: 8M, 35M, 150M, 650M, 3B and 15B.","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-d7e5c63feb0e0e627625","evidence-official-28fc17fbe0c1d99da219"],"source_locator":"ESM README: Pre-trained Models and Main models; official facebook/esm2_t33_650M_UR50D README licence metadata and config.json"},{"label":"Known versions","value":"ESM-2 checkpoint identifiers encode layer count, parameter scale and training-data tag.","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-d7e5c63feb0e0e627625","evidence-official-28fc17fbe0c1d99da219"],"source_locator":"ESM README: Pre-trained Models and Main models; official facebook/esm2_t33_650M_UR50D README licence metadata and config.json"},{"label":"Training data","value":"UniRef50 clusters with UniRef90 sampling; the pretrained-model table labels UR50/D 2021_04.","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-d7e5c63feb0e0e627625","evidence-official-28fc17fbe0c1d99da219"],"source_locator":"ESM README: Pre-trained Models and Main models; official facebook/esm2_t33_650M_UR50D README licence metadata and config.json"},{"label":"Training cutoff","value":"The pretrained-model table identifies training-data release UR50/D 2021_04; a corpus release date is not necessarily a last-deposited-sequence cutoff.","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-d7e5c63feb0e0e627625","evidence-official-28fc17fbe0c1d99da219"],"source_locator":"ESM README: Pre-trained Models and Main models; official facebook/esm2_t33_650M_UR50D README licence metadata and config.json"},{"label":"Context limits","value":"The checked esm2_t33_650M_UR50D configuration lists max_position_embeddings=1,026. This configuration field includes model positions and is not a claim of training or validated inference on 1,026 amino acids.","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-d7e5c63feb0e0e627625","evidence-official-28fc17fbe0c1d99da219"],"source_locator":"ESM README: Pre-trained Models and Main models; official facebook/esm2_t33_650M_UR50D README licence metadata and config.json"},{"label":"Weights licence","value":"The official facebook/esm2_t33_650M_UR50D model card declares MIT; this is the inspected checkpoint, not a licence inference from source code.","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-d7e5c63feb0e0e627625","evidence-official-28fc17fbe0c1d99da219"],"source_locator":"ESM README: Pre-trained Models and Main models; official facebook/esm2_t33_650M_UR50D README licence metadata and config.json"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/facebookresearch/esm","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-d7e5c63feb0e0e627625","evidence-official-28fc17fbe0c1d99da219"],"source_locator":"ESM README: Pre-trained Models and Main models; official facebook/esm2_t33_650M_UR50D README licence metadata and config.json"},{"label":"Code licence","value":"MIT","status":"source_checked","source_ids":["evidence-official-2e7c7649620407f50f6b"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"Released checkpoints span several sizes and can be used without constructing a multiple sequence alignment.","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-d7e5c63feb0e0e627625","evidence-official-28fc17fbe0c1d99da219"],"source_locator":"ESM README: Pre-trained Models and Main models; official facebook/esm2_t33_650M_UR50D README licence metadata and config.json"}],"limitations":[{"text":"A general embedding is not a directly measured function or structure. Fine-tuning, pooling and downstream heads remain part of each evaluated configuration.","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-d7e5c63feb0e0e627625","evidence-official-28fc17fbe0c1d99da219"],"source_locator":"ESM README: Pre-trained Models and Main models; official facebook/esm2_t33_650M_UR50D README licence metadata and config.json"}],"diagram":{"title":"ESM-2 workflow","steps":["Protein sequence","Transformer layers","Residue embeddings","Specified downstream analysis"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-d7e5c63feb0e0e627625","evidence-official-28fc17fbe0c1d99da219"],"source_locator":"ESM README: Pre-trained Models and Main models; official facebook/esm2_t33_650M_UR50D README licence metadata and config.json"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}},"description":"Protein sequence representation family","facets":{"areas":["protein-function"]},"id":"discovery-model-esm-2","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-proteingym"}],"name":"ESM-2","source_ids":["src-discovery-facebookresearch-esm"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-proteinbench"],"entity_level":"family","reported_name":"ESM-IF1","version":null,"historical_missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"ESM-IF1 designs protein sequences conditioned on backbone coordinates.","summary_source_ids":["evidence-official-29d4b229a3aa426a6dfb"],"summary_source_locator":"README.md: Inverse folding and Pre-trained Models","sections":[{"title":"How it works","body":"ESM-IF1 designs protein sequences conditioned on backbone coordinates. Geometric-vector-perceptron input processing followed by a sequence-to-sequence transformer. The documented inputs are protein backbone atom coordinates; the model supports missing backbone spans. The output consists of sampled protein sequences or conditional sequence likelihoods.","source_ids":["evidence-official-29d4b229a3aa426a6dfb"],"source_locator":"README.md: Inverse folding and Pre-trained Models"},{"title":"Versions and reproducibility","body":"esm_if1_gvp4_t16_142M_UR50. The applicable input limits require configuration-specific checking.","source_ids":["evidence-official-29d4b229a3aa426a6dfb"],"source_locator":"README.md: Inverse folding and Pre-trained Models"}],"facts":[{"label":"Model type","value":"Geometric encoder and inverse-folding transformer","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb"],"source_locator":"README.md: Inverse folding and Pre-trained Models"},{"label":"Architecture","value":"Geometric-vector-perceptron input processing followed by a sequence-to-sequence transformer.","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb"],"source_locator":"README.md: Inverse folding and Pre-trained Models"},{"label":"Inputs","value":"Protein backbone atom coordinates; the model supports missing backbone spans.","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb"],"source_locator":"README.md: Inverse folding and Pre-trained Models"},{"label":"Outputs","value":"Sampled protein sequences or conditional sequence likelihoods.","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb"],"source_locator":"README.md: Inverse folding and Pre-trained Models"},{"label":"Parameters","value":"The official checkpoint identifier contains 142M but the same repository model table states 124M. Both values are retained as a source discrepancy, without choosing a total.","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb"],"source_locator":"README.md: Inverse folding and Pre-trained Models"},{"label":"Known versions","value":"esm_if1_gvp4_t16_142M_UR50.","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb"],"source_locator":"README.md: Inverse folding and Pre-trained Models"},{"label":"Training data","value":"CATH 4.3 and predicted UniRef50 structures; README reports 12M structures predicted by AlphaFold2.","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb"],"source_locator":"README.md: Inverse folding and Pre-trained Models"},{"label":"Training cutoff","value":"The official model table identifies CATH 4.3 and predicted UniRef50 structures, but does not state a common latest-structure or sequence date.","status":"unreported","source_ids":["evidence-official-29d4b229a3aa426a6dfb"],"source_locator":"README.md: Inverse folding and Pre-trained Models"},{"label":"Context limits","value":"The reviewed inverse-folding usage and model table do not establish a universal maximum backbone length; the structural graph and selected inference configuration determine resource use.","status":"unreported","source_ids":["evidence-official-29d4b229a3aa426a6dfb"],"source_locator":"README.md: Inverse folding and Pre-trained Models"},{"label":"Weights licence","value":"Separate checkpoint-distribution terms are not stated in the inspected release documentation and licence material. The source-code licence alone is not recorded as an explicit weight grant.","status":"unreported","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-2e7c7649620407f50f6b"],"source_locator":"README.md: Inverse folding and Pre-trained Models; LICENSE: licence text"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/facebookresearch/esm","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb"],"source_locator":"README.md: Inverse folding and Pre-trained Models"},{"label":"Code licence","value":"MIT","status":"source_checked","source_ids":["evidence-official-2e7c7649620407f50f6b"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"Structure-conditioned design can use partially missing backbones because training included span masking.","source_ids":["evidence-official-29d4b229a3aa426a6dfb"],"source_locator":"README.md: Inverse folding and Pre-trained Models"}],"limitations":[{"text":"The public checkpoint name contains 142M but the README parameter table reports 124M. This discrepancy is retained rather than silently selecting a total.","source_ids":["evidence-official-29d4b229a3aa426a6dfb"],"source_locator":"README.md: Inverse folding and Pre-trained Models"}],"diagram":{"title":"ESM-IF1 workflow","steps":["Backbone coordinates","Geometric processing","Sequence transformer","Designed sequence or likelihood"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-29d4b229a3aa426a6dfb"],"source_locator":"README.md: Inverse folding and Pre-trained Models"},"coverage":"limited","gaps":["Training cutoff: The official model table identifies CATH 4.3 and predicted UniRef50 structures, but does not state a common latest-structure or sequence date.","Context limits: The reviewed inverse-folding usage and model table do not establish a universal maximum backbone length; the structural graph and selected inference configuration determine resource use.","Weights licence: Separate checkpoint-distribution terms are not stated in the inspected release documentation and licence material. The source-code licence alone is not recorded as an explicit weight grant."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}},"description":"Protein inverse folding model","facets":{"areas":["protein-structure"]},"id":"discovery-model-esm-if1","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-proteinbench"}],"name":"ESM-IF1","source_ids":["src-discovery-facebookresearch-esm"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-proteinbench"],"entity_level":"family","reported_name":"ESM3","version":null,"historical_missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"ESM3 generates and completes protein sequence, structure and functional annotations using a shared multimodal representation.","summary_source_ids":["evidence-official-4bf005200ebfb2586b78","evidence-official-0437be6150a3da9f8efc","evidence-official-1f027f8ecc53decdeec8"],"summary_source_locator":"_assets/ESM3_README.md: overview, ESM3 Family, Running Locally and Licenses","sections":[{"title":"How it works","body":"ESM3 generates and completes protein sequence, structure and functional annotations using a shared multimodal representation. Transformer generative masked-language model with discrete sequence, structure and function tracks; generation iteratively fills masked positions. The documented inputs are complete or partial protein sequence, structure and function-keyword tracks. The output consists of completed sequence, structure and functional tracks.","source_ids":["evidence-official-4bf005200ebfb2586b78","evidence-official-0437be6150a3da9f8efc","evidence-official-1f027f8ecc53decdeec8"],"source_locator":"_assets/ESM3_README.md: overview, ESM3 Family, Running Locally and Licenses"},{"title":"Versions and reproducibility","body":"Published March 2024 small/medium/large models, August 2024 small/medium models, and esm3-sm-open-v1 local weights. The applicable input limits require configuration-specific checking.","source_ids":["evidence-official-4bf005200ebfb2586b78","evidence-official-0437be6150a3da9f8efc","evidence-official-1f027f8ecc53decdeec8"],"source_locator":"_assets/ESM3_README.md: overview, ESM3 Family, Running Locally and Licenses"}],"facts":[{"label":"Model type","value":"Multitrack generative protein transformer","status":"source_checked","source_ids":["evidence-official-4bf005200ebfb2586b78","evidence-official-0437be6150a3da9f8efc","evidence-official-1f027f8ecc53decdeec8"],"source_locator":"_assets/ESM3_README.md: overview, ESM3 Family, Running Locally and Licenses"},{"label":"Architecture","value":"Transformer generative masked-language model with discrete sequence, structure and function tracks; generation iteratively fills masked positions.","status":"source_checked","source_ids":["evidence-official-4bf005200ebfb2586b78","evidence-official-0437be6150a3da9f8efc","evidence-official-1f027f8ecc53decdeec8"],"source_locator":"_assets/ESM3_README.md: overview, ESM3 Family, Running Locally and Licenses"},{"label":"Inputs","value":"Complete or partial protein sequence, structure and function-keyword tracks.","status":"source_checked","source_ids":["evidence-official-4bf005200ebfb2586b78","evidence-official-0437be6150a3da9f8efc","evidence-official-1f027f8ecc53decdeec8"],"source_locator":"_assets/ESM3_README.md: overview, ESM3 Family, Running Locally and Licenses"},{"label":"Outputs","value":"Completed sequence, structure and functional tracks.","status":"source_checked","source_ids":["evidence-official-4bf005200ebfb2586b78","evidence-official-0437be6150a3da9f8efc","evidence-official-1f027f8ecc53decdeec8"],"source_locator":"_assets/ESM3_README.md: overview, ESM3 Family, Running Locally and Licenses"},{"label":"Parameters","value":"1.4B small, 7B medium and 98B large variants.","status":"source_checked","source_ids":["evidence-official-4bf005200ebfb2586b78","evidence-official-0437be6150a3da9f8efc","evidence-official-1f027f8ecc53decdeec8"],"source_locator":"_assets/ESM3_README.md: overview, ESM3 Family, Running Locally and Licenses"},{"label":"Known versions","value":"Published March 2024 small/medium/large models, August 2024 small/medium models, and esm3-sm-open-v1 local weights.","status":"source_checked","source_ids":["evidence-official-4bf005200ebfb2586b78","evidence-official-0437be6150a3da9f8efc","evidence-official-1f027f8ecc53decdeec8"],"source_locator":"_assets/ESM3_README.md: overview, ESM3 Family, Running Locally and Licenses"},{"label":"Training data","value":"ESM3 overview reports 2.78 billion proteins and 771 billion unique tokens for the largest model; this is not a verified per-checkpoint manifest.","status":"source_checked","source_ids":["evidence-official-4bf005200ebfb2586b78","evidence-official-0437be6150a3da9f8efc","evidence-official-1f027f8ecc53decdeec8"],"source_locator":"_assets/ESM3_README.md: overview, ESM3 Family, Running Locally and Licenses"},{"label":"Training cutoff","value":"The inspected ESM3 release documentation does not supply one latest-data date shared across its sequence, structure and function tracks.","status":"unreported","source_ids":["evidence-official-4bf005200ebfb2586b78","evidence-official-0437be6150a3da9f8efc","evidence-official-1f027f8ecc53decdeec8"],"source_locator":"_assets/ESM3_README.md: overview, ESM3 Family, Running Locally and Licenses"},{"label":"Context limits","value":"The inspected ESM3 family documentation does not establish one context limit for local small, hosted medium and hosted large models; a specific service/checkpoint is required.","status":"unreported","source_ids":["evidence-official-4bf005200ebfb2586b78","evidence-official-0437be6150a3da9f8efc","evidence-official-1f027f8ecc53decdeec8"],"source_locator":"_assets/ESM3_README.md: overview, ESM3 Family, Running Locally and Licenses"},{"label":"Weights licence","value":"MIT stated in the inspected ESM3 README; model and API access conditions remain separate.","status":"source_checked","source_ids":["evidence-official-4bf005200ebfb2586b78","evidence-official-0437be6150a3da9f8efc","evidence-official-1f027f8ecc53decdeec8"],"source_locator":"_assets/ESM3_README.md: overview, ESM3 Family, Running Locally and Licenses"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/evolutionaryscale/esm","status":"source_checked","source_ids":["evidence-official-4bf005200ebfb2586b78","evidence-official-0437be6150a3da9f8efc","evidence-official-1f027f8ecc53decdeec8"],"source_locator":"_assets/ESM3_README.md: overview, ESM3 Family, Running Locally and Licenses"},{"label":"Code licence","value":"MIT","status":"source_checked","source_ids":["evidence-official-2f4711ecd64b0162e85b"],"source_locator":"LICENSE.md: licence text"}],"strengths":[{"text":"Partial prompts can constrain more than one biological property in the same generation workflow.","source_ids":["evidence-official-4bf005200ebfb2586b78","evidence-official-0437be6150a3da9f8efc","evidence-official-1f027f8ecc53decdeec8"],"source_locator":"_assets/ESM3_README.md: overview, ESM3 Family, Running Locally and Licenses"}],"limitations":[{"text":"Open local weights and API model names refer to different released sizes and dates; results require the actual configuration, not only the ESM3 family name.","source_ids":["evidence-official-4bf005200ebfb2586b78","evidence-official-0437be6150a3da9f8efc","evidence-official-1f027f8ecc53decdeec8"],"source_locator":"_assets/ESM3_README.md: overview, ESM3 Family, Running Locally and Licenses"}],"diagram":{"title":"ESM3 workflow","steps":["Partial protein tracks","Multitrack transformer","Iterative unmasking","Completed sequence or structure"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-4bf005200ebfb2586b78","evidence-official-0437be6150a3da9f8efc","evidence-official-1f027f8ecc53decdeec8"],"source_locator":"_assets/ESM3_README.md: overview, ESM3 Family, Running Locally and Licenses"},"coverage":"limited","gaps":["Training cutoff: The inspected ESM3 release documentation does not supply one latest-data date shared across its sequence, structure and function tracks.","Context limits: The inspected ESM3 family documentation does not establish one context limit for local small, hosted medium and hosted large models; a specific service/checkpoint is required."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}},"description":"Multimodal protein model family","facets":{"areas":["protein-structure"]},"id":"discovery-model-esm3","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-proteinbench"}],"name":"ESM3","source_ids":["src-discovery-evolutionaryscale-esm"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":[],"entity_level":"family","reported_name":"ESMC","version":null,"historical_missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"ESM C learns protein sequence representations for downstream analysis.","summary_source_ids":["evidence-official-4bf005200ebfb2586b78","evidence-official-0437be6150a3da9f8efc","evidence-official-1f027f8ecc53decdeec8","evidence-official-1d81a48c6dcc0e6dc14d","evidence-official-e69877a2a54345e162b3"],"summary_source_locator":"Official biohub/ESMC-6B card: Model Architecture, Parameters, Training Data, Training Procedure and Limitations; THIRD_PARTY_NOTICE.md","sections":[{"title":"How it works","body":"ESM C learns protein sequence representations for downstream analysis. Pre-normalized transformer with rotary embeddings, SwiGLU feed-forward activations and no linear/layer-norm biases. The documented inputs are protein amino-acid sequences. The output consists of final-layer or all-layer protein representations and masked-token outputs.","source_ids":["evidence-official-4bf005200ebfb2586b78","evidence-official-0437be6150a3da9f8efc","evidence-official-1f027f8ecc53decdeec8","evidence-official-1d81a48c6dcc0e6dc14d","evidence-official-e69877a2a54345e162b3"],"source_locator":"Official biohub/ESMC-6B card: Model Architecture, Parameters, Training Data, Training Procedure and Limitations; THIRD_PARTY_NOTICE.md"},{"title":"Versions and reproducibility","body":"esmc-600m-2024-12 API identifier and biohub/ESMC-6B local example. The card specifies a2,048-token window after an initial 512-token training phase.","source_ids":["evidence-official-4bf005200ebfb2586b78","evidence-official-0437be6150a3da9f8efc","evidence-official-1f027f8ecc53decdeec8","evidence-official-1d81a48c6dcc0e6dc14d","evidence-official-e69877a2a54345e162b3"],"source_locator":"Official biohub/ESMC-6B card: Model Architecture, Parameters, Training Data, Training Procedure and Limitations; THIRD_PARTY_NOTICE.md"}],"facts":[{"label":"Model type","value":"Protein representation transformer","status":"source_checked","source_ids":["evidence-official-4bf005200ebfb2586b78","evidence-official-0437be6150a3da9f8efc","evidence-official-1f027f8ecc53decdeec8","evidence-official-1d81a48c6dcc0e6dc14d","evidence-official-e69877a2a54345e162b3"],"source_locator":"Official biohub/ESMC-6B card: Model Architecture, Parameters, Training Data, Training Procedure and Limitations; THIRD_PARTY_NOTICE.md"},{"label":"Architecture","value":"Pre-normalized transformer with rotary embeddings, SwiGLU feed-forward activations and no linear/layer-norm biases.","status":"source_checked","source_ids":["evidence-official-4bf005200ebfb2586b78","evidence-official-0437be6150a3da9f8efc","evidence-official-1f027f8ecc53decdeec8","evidence-official-1d81a48c6dcc0e6dc14d","evidence-official-e69877a2a54345e162b3"],"source_locator":"Official biohub/ESMC-6B card: Model Architecture, Parameters, Training Data, Training Procedure and Limitations; THIRD_PARTY_NOTICE.md"},{"label":"Inputs","value":"Protein amino-acid sequences.","status":"source_checked","source_ids":["evidence-official-4bf005200ebfb2586b78","evidence-official-0437be6150a3da9f8efc","evidence-official-1f027f8ecc53decdeec8","evidence-official-1d81a48c6dcc0e6dc14d","evidence-official-e69877a2a54345e162b3"],"source_locator":"Official biohub/ESMC-6B card: Model Architecture, Parameters, Training Data, Training Procedure and Limitations; THIRD_PARTY_NOTICE.md"},{"label":"Outputs","value":"Final-layer or all-layer protein representations and masked-token outputs.","status":"source_checked","source_ids":["evidence-official-4bf005200ebfb2586b78","evidence-official-0437be6150a3da9f8efc","evidence-official-1f027f8ecc53decdeec8","evidence-official-1d81a48c6dcc0e6dc14d","evidence-official-e69877a2a54345e162b3"],"source_locator":"Official biohub/ESMC-6B card: Model Architecture, Parameters, Training Data, Training Procedure and Limitations; THIRD_PARTY_NOTICE.md"},{"label":"Parameters","value":"300M/30 layers, 600M/36 layers and 6B/80 layers in the checked model card.","status":"source_checked","source_ids":["evidence-official-4bf005200ebfb2586b78","evidence-official-0437be6150a3da9f8efc","evidence-official-1f027f8ecc53decdeec8","evidence-official-1d81a48c6dcc0e6dc14d","evidence-official-e69877a2a54345e162b3"],"source_locator":"Official biohub/ESMC-6B card: Model Architecture, Parameters, Training Data, Training Procedure and Limitations; THIRD_PARTY_NOTICE.md"},{"label":"Known versions","value":"esmc-600m-2024-12 API identifier and biohub/ESMC-6B local example.","status":"source_checked","source_ids":["evidence-official-4bf005200ebfb2586b78","evidence-official-0437be6150a3da9f8efc","evidence-official-1f027f8ecc53decdeec8","evidence-official-1d81a48c6dcc0e6dc14d","evidence-official-e69877a2a54345e162b3"],"source_locator":"Official biohub/ESMC-6B card: Model Architecture, Parameters, Training Data, Training Procedure and Limitations; THIRD_PARTY_NOTICE.md"},{"label":"Training data","value":"UniRef, MGnify and JGI protein sequences clustered at 70% identity. The card distinguishes83M, 372M and 2B clusters respectively from the number of repeated training tokens.","status":"source_checked","source_ids":["evidence-official-4bf005200ebfb2586b78","evidence-official-0437be6150a3da9f8efc","evidence-official-1f027f8ecc53decdeec8","evidence-official-1d81a48c6dcc0e6dc14d","evidence-official-e69877a2a54345e162b3"],"source_locator":"Official biohub/ESMC-6B card: Model Architecture, Parameters, Training Data, Training Procedure and Limitations; THIRD_PARTY_NOTICE.md"},{"label":"Training cutoff","value":"The official ESMC-6B card identifies UniRef, MGnify and JGI clusters and training stages, but does not give a latest-sequence date shared across those corpora.","status":"unreported","source_ids":["evidence-official-4bf005200ebfb2586b78","evidence-official-0437be6150a3da9f8efc","evidence-official-1f027f8ecc53decdeec8","evidence-official-1d81a48c6dcc0e6dc14d","evidence-official-e69877a2a54345e162b3"],"source_locator":"Official biohub/ESMC-6B card: Model Architecture, Parameters, Training Data, Training Procedure and Limitations; THIRD_PARTY_NOTICE.md"},{"label":"Context limits","value":"The card specifies a2,048-token window after an initial 512-token training phase.","status":"source_checked","source_ids":["evidence-official-4bf005200ebfb2586b78","evidence-official-0437be6150a3da9f8efc","evidence-official-1f027f8ecc53decdeec8","evidence-official-1d81a48c6dcc0e6dc14d","evidence-official-e69877a2a54345e162b3"],"source_locator":"Official biohub/ESMC-6B card: Model Architecture, Parameters, Training Data, Training Procedure and Limitations; THIRD_PARTY_NOTICE.md"},{"label":"Weights licence","value":"MIT is declared alongside third-party notices in the checked6B card; THIRD_PARTY_NOTICE.md lists dependency licences.","status":"source_checked","source_ids":["evidence-official-4bf005200ebfb2586b78","evidence-official-0437be6150a3da9f8efc","evidence-official-1f027f8ecc53decdeec8","evidence-official-1d81a48c6dcc0e6dc14d","evidence-official-e69877a2a54345e162b3"],"source_locator":"Official biohub/ESMC-6B card: Model Architecture, Parameters, Training Data, Training Procedure and Limitations; THIRD_PARTY_NOTICE.md"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/evolutionaryscale/esm","status":"source_checked","source_ids":["evidence-official-4bf005200ebfb2586b78","evidence-official-0437be6150a3da9f8efc","evidence-official-1f027f8ecc53decdeec8","evidence-official-1d81a48c6dcc0e6dc14d","evidence-official-e69877a2a54345e162b3"],"source_locator":"Official biohub/ESMC-6B card: Model Architecture, Parameters, Training Data, Training Procedure and Limitations; THIRD_PARTY_NOTICE.md"},{"label":"Code licence","value":"MIT","status":"source_checked","source_ids":["evidence-official-2f4711ecd64b0162e85b"],"source_locator":"LICENSE.md: licence text"}],"strengths":[{"text":"The implementation supports local inference and a hosted interface, with downloadable model variants.","source_ids":["evidence-official-4bf005200ebfb2586b78","evidence-official-0437be6150a3da9f8efc","evidence-official-1f027f8ecc53decdeec8","evidence-official-1d81a48c6dcc0e6dc14d","evidence-official-e69877a2a54345e162b3"],"source_locator":"Official biohub/ESMC-6B card: Model Architecture, Parameters, Training Data, Training Procedure and Limitations; THIRD_PARTY_NOTICE.md"}],"limitations":[{"text":"Hosted API names and downloadable weights identify different versions. The main repository does not by itself provide a complete per-checkpoint training manifest.","source_ids":["evidence-official-4bf005200ebfb2586b78","evidence-official-0437be6150a3da9f8efc","evidence-official-1f027f8ecc53decdeec8","evidence-official-1d81a48c6dcc0e6dc14d","evidence-official-e69877a2a54345e162b3"],"source_locator":"Official biohub/ESMC-6B card: Model Architecture, Parameters, Training Data, Training Procedure and Limitations; THIRD_PARTY_NOTICE.md"}],"diagram":{"title":"ESMC workflow","steps":["Protein sequence","ESM C transformer","Layer representations","Downstream analysis"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-4bf005200ebfb2586b78","evidence-official-0437be6150a3da9f8efc","evidence-official-1f027f8ecc53decdeec8","evidence-official-1d81a48c6dcc0e6dc14d","evidence-official-e69877a2a54345e162b3"],"source_locator":"Official biohub/ESMC-6B card: Model Architecture, Parameters, Training Data, Training Procedure and Limitations; THIRD_PARTY_NOTICE.md"},"coverage":"limited","gaps":["Training cutoff: The official ESMC-6B card identifies UniRef, MGnify and JGI clusters and training stages, but does not give a latest-sequence date shared across those corpora."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}},"description":"Protein representation family","facets":{"areas":["protein-function"]},"id":"discovery-model-esmc","kind":"model","links":[],"name":"ESMC","source_ids":["src-discovery-evolutionaryscale-esm"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-proteinbench"],"entity_level":"family","reported_name":"ESMFold","version":null,"historical_missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"ESMFold predicts protein structures directly from amino-acid sequence using ESM-2 representations.","summary_source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-fa53ab621979fb5f37c8","evidence-official-867f2d1dcb4084248daa"],"summary_source_locator":"ESM README: ESMFold Structure Prediction; official facebook/esmfold_v1 model card and config.json esmfold_config","sections":[{"title":"How it works","body":"ESMFold predicts protein structures directly from amino-acid sequence using ESM-2 representations. ESM-2 sequence representations feed a folding trunk and structure module. The checked v1 configuration has 48 trunk blocks, eight structure-module blocks and up to four recycles. The documented inputs are protein amino-acid sequence; the ESMFold interface also accepts chains separated by a colon. The output consists of predicted PDB structure and confidence values.","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-fa53ab621979fb5f37c8","evidence-official-867f2d1dcb4084248daa"],"source_locator":"ESM README: ESMFold Structure Prediction; official facebook/esmfold_v1 model card and config.json esmfold_config"},{"title":"Versions and reproducibility","body":"esmfold_v0 and esmfold_v1; v1 is the repository recommendation. Inference length is constrained by memory; the repository documents chunking and CPU offload. Backbone position settings do not alone establish the full folding pipeline limit.","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-fa53ab621979fb5f37c8","evidence-official-867f2d1dcb4084248daa"],"source_locator":"ESM README: ESMFold Structure Prediction; official facebook/esmfold_v1 model card and config.json esmfold_config"}],"facts":[{"label":"Model type","value":"Sequence-to-structure protein prediction pipeline","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-fa53ab621979fb5f37c8","evidence-official-867f2d1dcb4084248daa"],"source_locator":"ESM README: ESMFold Structure Prediction; official facebook/esmfold_v1 model card and config.json esmfold_config"},{"label":"Architecture","value":"ESM-2 sequence representations feed a folding trunk and structure module. The checked v1 configuration has 48 trunk blocks, eight structure-module blocks and up to four recycles.","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-fa53ab621979fb5f37c8","evidence-official-867f2d1dcb4084248daa"],"source_locator":"ESM README: ESMFold Structure Prediction; official facebook/esmfold_v1 model card and config.json esmfold_config"},{"label":"Inputs","value":"Protein amino-acid sequence; the ESMFold interface also accepts chains separated by a colon.","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-fa53ab621979fb5f37c8","evidence-official-867f2d1dcb4084248daa"],"source_locator":"ESM README: ESMFold Structure Prediction; official facebook/esmfold_v1 model card and config.json esmfold_config"},{"label":"Outputs","value":"Predicted PDB structure and confidence values.","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-fa53ab621979fb5f37c8","evidence-official-867f2d1dcb4084248daa"],"source_locator":"ESM README: ESMFold Structure Prediction; official facebook/esmfold_v1 model card and config.json esmfold_config"},{"label":"Parameters","value":"The released v1 configuration identifies an ESM-2 3B backbone plus a folding trunk and structure module. The 3B figure is not the total size of the complete predictor.","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-fa53ab621979fb5f37c8","evidence-official-867f2d1dcb4084248daa"],"source_locator":"ESM README: ESMFold Structure Prediction; official facebook/esmfold_v1 model card and config.json esmfold_config"},{"label":"Known versions","value":"esmfold_v0 and esmfold_v1; v1 is the repository recommendation.","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-fa53ab621979fb5f37c8","evidence-official-867f2d1dcb4084248daa"],"source_locator":"ESM README: ESMFold Structure Prediction; official facebook/esmfold_v1 model card and config.json esmfold_config"},{"label":"Training data","value":"PDB and UniRef50 are listed for ESMFold. Full structural training-cutoff verification remains outstanding.","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-fa53ab621979fb5f37c8","evidence-official-867f2d1dcb4084248daa"],"source_locator":"ESM README: ESMFold Structure Prediction; official facebook/esmfold_v1 model card and config.json esmfold_config"},{"label":"Training cutoff","value":"PDB and UniRef50 are identified in the official model table; the inspected ESMFold-v1 card and configuration do not supply a shared latest-data date for both components.","status":"unreported","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-fa53ab621979fb5f37c8","evidence-official-867f2d1dcb4084248daa"],"source_locator":"ESM README: ESMFold Structure Prediction; official facebook/esmfold_v1 model card and config.json esmfold_config"},{"label":"Context limits","value":"Inference length is constrained by memory; the repository documents chunking and CPU offload. Backbone position settings do not alone establish the full folding pipeline limit.","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-fa53ab621979fb5f37c8","evidence-official-867f2d1dcb4084248daa"],"source_locator":"ESM README: ESMFold Structure Prediction; official facebook/esmfold_v1 model card and config.json esmfold_config"},{"label":"Weights licence","value":"MIT declared by the official facebook/esmfold_v1 model card.","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-fa53ab621979fb5f37c8","evidence-official-867f2d1dcb4084248daa"],"source_locator":"ESM README: ESMFold Structure Prediction; official facebook/esmfold_v1 model card and config.json esmfold_config"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/facebookresearch/esm","status":"source_checked","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-fa53ab621979fb5f37c8","evidence-official-867f2d1dcb4084248daa"],"source_locator":"ESM README: ESMFold Structure Prediction; official facebook/esmfold_v1 model card and config.json esmfold_config"},{"label":"Code licence","value":"MIT","status":"source_checked","source_ids":["evidence-official-2e7c7649620407f50f6b"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"Supports sequence-only structure prediction without an MSA search.","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-fa53ab621979fb5f37c8","evidence-official-867f2d1dcb4084248daa"],"source_locator":"ESM README: ESMFold Structure Prediction; official facebook/esmfold_v1 model card and config.json esmfold_config"}],"limitations":[{"text":"ESMFold v0 and v1 are different releases. The repository discourages using structure-module-only ablation models as the standard predictor.","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-fa53ab621979fb5f37c8","evidence-official-867f2d1dcb4084248daa"],"source_locator":"ESM README: ESMFold Structure Prediction; official facebook/esmfold_v1 model card and config.json esmfold_config"}],"diagram":{"title":"ESMFold workflow","steps":["Protein sequence","ESM-2 representations","Folding module","Predicted structure and confidence"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-29d4b229a3aa426a6dfb","evidence-official-fa53ab621979fb5f37c8","evidence-official-867f2d1dcb4084248daa"],"source_locator":"ESM README: ESMFold Structure Prediction; official facebook/esmfold_v1 model card and config.json esmfold_config"},"coverage":"limited","gaps":["Training cutoff: PDB and UniRef50 are identified in the official model table; the inspected ESMFold-v1 card and configuration do not supply a shared latest-data date for both components."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}},"description":"Protein structure prediction model","facets":{"areas":["protein-structure"]},"id":"discovery-model-esmfold","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-proteinbench"}],"name":"ESMFold","source_ids":["src-discovery-facebookresearch-esm"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":[],"entity_level":"family","reported_name":"ESMFold2","version":null,"historical_missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"ESMFold2 predicts biomolecular structures from protein, DNA, RNA and ligand inputs, optionally using protein alignments.","summary_source_ids":["evidence-official-4bf005200ebfb2586b78","evidence-official-0437be6150a3da9f8efc","evidence-official-1f027f8ecc53decdeec8","evidence-official-d65bf8ba587b6841e0d9","evidence-official-6a331ce74fa7d11e0bfe"],"summary_source_locator":"Official biohub/ESMFold2 card: Model Details, Model Variants, Training Data and Biases and Limitations; official source README and THIRD_PARTY_NOTICE.md","sections":[{"title":"How it works","body":"ESMFold2 predicts biomolecular structures from protein, DNA, RNA and ligand inputs, optionally using protein alignments. ESM C 6B embeddings coupled to a diffusion-based structure prediction architecture. The documented inputs are protein, DNA/RNA, modified-residue and small-molecule specifications; optional protein MSAs for the full variant. The output consists of all-atom complex coordinates with confidence estimates and optional distogram predictions.","source_ids":["evidence-official-4bf005200ebfb2586b78","evidence-official-0437be6150a3da9f8efc","evidence-official-1f027f8ecc53decdeec8","evidence-official-d65bf8ba587b6841e0d9","evidence-official-6a331ce74fa7d11e0bfe"],"source_locator":"Official biohub/ESMFold2 card: Model Details, Model Variants, Training Data and Biases and Limitations; official source README and THIRD_PARTY_NOTICE.md"},{"title":"Versions and reproducibility","body":"ESMFold2 supports optional MSA conditioning; ESMFold2-Fast is a distinct single-sequence inference variant. The applicable input limits require configuration-specific checking.","source_ids":["evidence-official-4bf005200ebfb2586b78","evidence-official-0437be6150a3da9f8efc","evidence-official-1f027f8ecc53decdeec8","evidence-official-d65bf8ba587b6841e0d9","evidence-official-6a331ce74fa7d11e0bfe"],"source_locator":"Official biohub/ESMFold2 card: Model Details, Model Variants, Training Data and Biases and Limitations; official source README and THIRD_PARTY_NOTICE.md"}],"facts":[{"label":"Model type","value":"Biomolecular diffusion structure predictor with protein language-model features","status":"source_checked","source_ids":["evidence-official-4bf005200ebfb2586b78","evidence-official-0437be6150a3da9f8efc","evidence-official-1f027f8ecc53decdeec8","evidence-official-d65bf8ba587b6841e0d9","evidence-official-6a331ce74fa7d11e0bfe"],"source_locator":"Official biohub/ESMFold2 card: Model Details, Model Variants, Training Data and Biases and Limitations; official source README and THIRD_PARTY_NOTICE.md"},{"label":"Architecture","value":"ESM C 6B embeddings coupled to a diffusion-based structure prediction architecture.","status":"source_checked","source_ids":["evidence-official-4bf005200ebfb2586b78","evidence-official-0437be6150a3da9f8efc","evidence-official-1f027f8ecc53decdeec8","evidence-official-d65bf8ba587b6841e0d9","evidence-official-6a331ce74fa7d11e0bfe"],"source_locator":"Official biohub/ESMFold2 card: Model Details, Model Variants, Training Data and Biases and Limitations; official source README and THIRD_PARTY_NOTICE.md"},{"label":"Inputs","value":"Protein, DNA/RNA, modified-residue and small-molecule specifications; optional protein MSAs for the full variant.","status":"source_checked","source_ids":["evidence-official-4bf005200ebfb2586b78","evidence-official-0437be6150a3da9f8efc","evidence-official-1f027f8ecc53decdeec8","evidence-official-d65bf8ba587b6841e0d9","evidence-official-6a331ce74fa7d11e0bfe"],"source_locator":"Official biohub/ESMFold2 card: Model Details, Model Variants, Training Data and Biases and Limitations; official source README and THIRD_PARTY_NOTICE.md"},{"label":"Outputs","value":"All-atom complex coordinates with confidence estimates and optional distogram predictions.","status":"source_checked","source_ids":["evidence-official-4bf005200ebfb2586b78","evidence-official-0437be6150a3da9f8efc","evidence-official-1f027f8ecc53decdeec8","evidence-official-d65bf8ba587b6841e0d9","evidence-official-6a331ce74fa7d11e0bfe"],"source_locator":"Official biohub/ESMFold2 card: Model Details, Model Variants, Training Data and Biases and Limitations; official source README and THIRD_PARTY_NOTICE.md"},{"label":"Parameters","value":"The official card identifies a 6B ESM C backbone. It does not state a total including the complete molecular structure and confidence modules.","status":"unreported","source_ids":["evidence-official-4bf005200ebfb2586b78","evidence-official-0437be6150a3da9f8efc","evidence-official-1f027f8ecc53decdeec8","evidence-official-d65bf8ba587b6841e0d9","evidence-official-6a331ce74fa7d11e0bfe"],"source_locator":"Official biohub/ESMFold2 card: Model Details, Model Variants, Training Data and Biases and Limitations; official source README and THIRD_PARTY_NOTICE.md"},{"label":"Known versions","value":"ESMFold2 supports optional MSA conditioning; ESMFold2-Fast is a distinct single-sequence inference variant.","status":"source_checked","source_ids":["evidence-official-4bf005200ebfb2586b78","evidence-official-0437be6150a3da9f8efc","evidence-official-1f027f8ecc53decdeec8","evidence-official-d65bf8ba587b6841e0d9","evidence-official-6a331ce74fa7d11e0bfe"],"source_locator":"Official biohub/ESMFold2 card: Model Details, Model Variants, Training Data and Biases and Limitations; official source README and THIRD_PARTY_NOTICE.md"},{"label":"Training data","value":"The checked model card identifies PDB structures and AlphaFoldDB-derived training data.","status":"source_checked","source_ids":["evidence-official-4bf005200ebfb2586b78","evidence-official-0437be6150a3da9f8efc","evidence-official-1f027f8ecc53decdeec8","evidence-official-d65bf8ba587b6841e0d9","evidence-official-6a331ce74fa7d11e0bfe"],"source_locator":"Official biohub/ESMFold2 card: Model Details, Model Variants, Training Data and Biases and Limitations; official source README and THIRD_PARTY_NOTICE.md"},{"label":"Training cutoff","value":"September 2021 is the model-card data cutoff for both ESMFold2 and ESMFold2-Fast; the separate ESMC backbone corpus has its own provenance.","status":"source_checked","source_ids":["evidence-official-4bf005200ebfb2586b78","evidence-official-0437be6150a3da9f8efc","evidence-official-1f027f8ecc53decdeec8","evidence-official-d65bf8ba587b6841e0d9","evidence-official-6a331ce74fa7d11e0bfe"],"source_locator":"Official biohub/ESMFold2 card: Model Details, Model Variants, Training Data and Biases and Limitations; official source README and THIRD_PARTY_NOTICE.md"},{"label":"Context limits","value":"The inspected card describes Full and Fast variants without giving a single maximum token/atom budget that applies to every supported molecular composition.","status":"unreported","source_ids":["evidence-official-4bf005200ebfb2586b78","evidence-official-0437be6150a3da9f8efc","evidence-official-1f027f8ecc53decdeec8","evidence-official-d65bf8ba587b6841e0d9","evidence-official-6a331ce74fa7d11e0bfe"],"source_locator":"Official biohub/ESMFold2 card: Model Details, Model Variants, Training Data and Biases and Limitations; official source README and THIRD_PARTY_NOTICE.md"},{"label":"Weights licence","value":"MIT declared in the official ESMFold2 card, with linked third-party dependency notices.","status":"source_checked","source_ids":["evidence-official-4bf005200ebfb2586b78","evidence-official-0437be6150a3da9f8efc","evidence-official-1f027f8ecc53decdeec8","evidence-official-d65bf8ba587b6841e0d9","evidence-official-6a331ce74fa7d11e0bfe"],"source_locator":"Official biohub/ESMFold2 card: Model Details, Model Variants, Training Data and Biases and Limitations; official source README and THIRD_PARTY_NOTICE.md"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/evolutionaryscale/esm","status":"source_checked","source_ids":["evidence-official-4bf005200ebfb2586b78","evidence-official-0437be6150a3da9f8efc","evidence-official-1f027f8ecc53decdeec8","evidence-official-d65bf8ba587b6841e0d9","evidence-official-6a331ce74fa7d11e0bfe"],"source_locator":"Official biohub/ESMFold2 card: Model Details, Model Variants, Training Data and Biases and Limitations; official source README and THIRD_PARTY_NOTICE.md"},{"label":"Code licence","value":"MIT","status":"source_checked","source_ids":["evidence-official-2f4711ecd64b0162e85b"],"source_locator":"LICENSE.md: licence text"}],"strengths":[{"text":"Supports both sequence-only prediction and an MSA-assisted mode.","source_ids":["evidence-official-4bf005200ebfb2586b78","evidence-official-0437be6150a3da9f8efc","evidence-official-1f027f8ecc53decdeec8","evidence-official-d65bf8ba587b6841e0d9","evidence-official-6a331ce74fa7d11e0bfe"],"source_locator":"Official biohub/ESMFold2 card: Model Details, Model Variants, Training Data and Biases and Limitations; official source README and THIRD_PARTY_NOTICE.md"}],"limitations":[{"text":"The 6B figure describes the ESM C backbone, not a verified parameter total for the complete folding system. Sequence-only and MSA-assisted runs need separate evaluation identities.","source_ids":["evidence-official-4bf005200ebfb2586b78","evidence-official-0437be6150a3da9f8efc","evidence-official-1f027f8ecc53decdeec8","evidence-official-d65bf8ba587b6841e0d9","evidence-official-6a331ce74fa7d11e0bfe"],"source_locator":"Official biohub/ESMFold2 card: Model Details, Model Variants, Training Data and Biases and Limitations; official source README and THIRD_PARTY_NOTICE.md"}],"diagram":{"title":"ESMFold2 workflow","steps":["Protein, DNA, RNA and ligands","Molecular features; ESM C for proteins","Optional protein MSA conditioning","Diffusion structure prediction","Complex coordinates and confidence"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-4bf005200ebfb2586b78","evidence-official-0437be6150a3da9f8efc","evidence-official-1f027f8ecc53decdeec8","evidence-official-d65bf8ba587b6841e0d9","evidence-official-6a331ce74fa7d11e0bfe"],"source_locator":"Official biohub/ESMFold2 card: Model Details, Model Variants, Training Data and Biases and Limitations; official source README and THIRD_PARTY_NOTICE.md"},"coverage":"limited","gaps":["Parameters: The official card identifies a 6B ESM C backbone. It does not state a total including the complete molecular structure and confidence modules.","Context limits: The inspected card describes Full and Fast variants without giving a single maximum token/atom budget that applies to every supported molecular composition."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}},"description":"Protein structure and complex prediction","facets":{"areas":["protein-structure"]},"id":"discovery-model-esmfold2","kind":"model","links":[],"name":"ESMFold2","source_ids":["src-discovery-evolutionaryscale-esm"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-geneb"],"entity_level":"family","reported_name":"Evo 2","version":null,"historical_missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"Evo 2 models and generates DNA over long contexts at single-nucleotide resolution.","summary_source_ids":["evidence-official-a248990b44d481814f47","evidence-official-849b751ed4bf88622749","evidence-official-b8df803e8f55324a4eb5","evidence-official-8717ffb993cc8f7eddec"],"summary_source_locator":"Evo 2 paper: architecture, training, and data, Figure 1 and Discussion; official checkpoints table and evo2_7b licence metadata","sections":[{"title":"How it works","body":"Evo 2 models and generates DNA over long contexts at single-nucleotide resolution. StripedHyena 2 hybrid architecture combining short, medium and long convolution operators with attention, trained autoregressively at single-base resolution. The documented inputs are DNA sequences represented at single-base resolution. The output consists of next-token outputs, embeddings and generated DNA sequences.","source_ids":["evidence-official-a248990b44d481814f47","evidence-official-849b751ed4bf88622749","evidence-official-b8df803e8f55324a4eb5","evidence-official-8717ffb993cc8f7eddec"],"source_locator":"Evo 2 paper: architecture, training, and data, Figure 1 and Discussion; official checkpoints table and evo2_7b licence metadata"},{"title":"Versions and reproducibility","body":"Base 8K models, long-context 1M models, 7B 262K model and separately fine-tuned Microviridae model. Checkpoint-dependent: 8K, 262K or 1M bases as listed in the Checkpoints table.","source_ids":["evidence-official-a248990b44d481814f47","evidence-official-849b751ed4bf88622749","evidence-official-b8df803e8f55324a4eb5","evidence-official-8717ffb993cc8f7eddec"],"source_locator":"Evo 2 paper: architecture, training, and data, Figure 1 and Discussion; official checkpoints table and evo2_7b licence metadata"}],"facts":[{"label":"Model type","value":"Autoregressive DNA model with StripedHyena 2","status":"source_checked","source_ids":["evidence-official-a248990b44d481814f47","evidence-official-849b751ed4bf88622749","evidence-official-b8df803e8f55324a4eb5","evidence-official-8717ffb993cc8f7eddec"],"source_locator":"Evo 2 paper: architecture, training, and data, Figure 1 and Discussion; official checkpoints table and evo2_7b licence metadata"},{"label":"Architecture","value":"StripedHyena 2 hybrid architecture combining short, medium and long convolution operators with attention, trained autoregressively at single-base resolution.","status":"source_checked","source_ids":["evidence-official-a248990b44d481814f47","evidence-official-849b751ed4bf88622749","evidence-official-b8df803e8f55324a4eb5","evidence-official-8717ffb993cc8f7eddec"],"source_locator":"Evo 2 paper: architecture, training, and data, Figure 1 and Discussion; official checkpoints table and evo2_7b licence metadata"},{"label":"Inputs","value":"DNA sequences represented at single-base resolution.","status":"source_checked","source_ids":["evidence-official-a248990b44d481814f47","evidence-official-849b751ed4bf88622749","evidence-official-b8df803e8f55324a4eb5","evidence-official-8717ffb993cc8f7eddec"],"source_locator":"Evo 2 paper: architecture, training, and data, Figure 1 and Discussion; official checkpoints table and evo2_7b licence metadata"},{"label":"Outputs","value":"Next-token outputs, embeddings and generated DNA sequences.","status":"source_checked","source_ids":["evidence-official-a248990b44d481814f47","evidence-official-849b751ed4bf88622749","evidence-official-b8df803e8f55324a4eb5","evidence-official-8717ffb993cc8f7eddec"],"source_locator":"Evo 2 paper: architecture, training, and data, Figure 1 and Discussion; official checkpoints table and evo2_7b licence metadata"},{"label":"Parameters","value":"1B, 7B, 20B and 40B checkpoints are listed.","status":"source_checked","source_ids":["evidence-official-a248990b44d481814f47","evidence-official-849b751ed4bf88622749","evidence-official-b8df803e8f55324a4eb5","evidence-official-8717ffb993cc8f7eddec"],"source_locator":"Evo 2 paper: architecture, training, and data, Figure 1 and Discussion; official checkpoints table and evo2_7b licence metadata"},{"label":"Known versions","value":"Base 8K models, long-context 1M models, 7B 262K model and separately fine-tuned Microviridae model.","status":"source_checked","source_ids":["evidence-official-a248990b44d481814f47","evidence-official-849b751ed4bf88622749","evidence-official-b8df803e8f55324a4eb5","evidence-official-8717ffb993cc8f7eddec"],"source_locator":"Evo 2 paper: architecture, training, and data, Figure 1 and Discussion; official checkpoints table and evo2_7b licence metadata"},{"label":"Training data","value":"OpenGenome2 contains more than 8.8T curated nucleotides across bacteria, archaea, eukaryotes and bacteriophage. The paper separates 2.4T tokens of training exposure for 7B from 9.3T for 40B; eukaryotic-host viral sequences were excluded.","status":"source_checked","source_ids":["evidence-official-a248990b44d481814f47","evidence-official-849b751ed4bf88622749","evidence-official-b8df803e8f55324a4eb5","evidence-official-8717ffb993cc8f7eddec"],"source_locator":"Evo 2 paper: architecture, training, and data, Figure 1 and Discussion; official checkpoints table and evo2_7b licence metadata"},{"label":"Training cutoff","value":"OpenGenome2 combines multiple nucleotide collections. The inspected paper and released checkpoint documentation do not provide one latest-deposition date that covers every component.","status":"unreported","source_ids":["evidence-official-a248990b44d481814f47","evidence-official-849b751ed4bf88622749","evidence-official-b8df803e8f55324a4eb5","evidence-official-8717ffb993cc8f7eddec"],"source_locator":"Evo 2 paper: architecture, training, and data, Figure 1 and Discussion; official checkpoints table and evo2_7b licence metadata"},{"label":"Context limits","value":"Checkpoint-dependent: 8K, 262K or 1M bases as listed in the Checkpoints table.","status":"source_checked","source_ids":["evidence-official-a248990b44d481814f47","evidence-official-849b751ed4bf88622749","evidence-official-b8df803e8f55324a4eb5","evidence-official-8717ffb993cc8f7eddec"],"source_locator":"Evo 2 paper: architecture, training, and data, Figure 1 and Discussion; official checkpoints table and evo2_7b licence metadata"},{"label":"Weights licence","value":"Apache-2.0 is declared in the inspected ArcInstitute/evo2_7b model card; other checkpoints require their own pinned terms.","status":"source_checked","source_ids":["evidence-official-a248990b44d481814f47","evidence-official-849b751ed4bf88622749","evidence-official-b8df803e8f55324a4eb5","evidence-official-8717ffb993cc8f7eddec"],"source_locator":"Evo 2 paper: architecture, training, and data, Figure 1 and Discussion; official checkpoints table and evo2_7b licence metadata"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/ArcInstitute/evo2","status":"source_checked","source_ids":["evidence-official-a248990b44d481814f47","evidence-official-849b751ed4bf88622749","evidence-official-b8df803e8f55324a4eb5","evidence-official-8717ffb993cc8f7eddec"],"source_locator":"Evo 2 paper: architecture, training, and data, Figure 1 and Discussion; official checkpoints table and evo2_7b licence metadata"},{"label":"Code licence","value":"Apache-2.0","status":"source_checked","source_ids":["evidence-official-67ee1cc31060ba8c9569"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"Different released context lengths and model scales support a range of sequence modeling workflows.","source_ids":["evidence-official-a248990b44d481814f47","evidence-official-849b751ed4bf88622749","evidence-official-b8df803e8f55324a4eb5","evidence-official-8717ffb993cc8f7eddec"],"source_locator":"Evo 2 paper: architecture, training, and data, Figure 1 and Discussion; official checkpoints table and evo2_7b licence metadata"}],"limitations":[{"text":"Hardware requirements differ by checkpoint: the README requires FP8/Transformer Engine and Hopper GPUs for some scales, while 7B supports bfloat 16 on a wider set of GPUs.","source_ids":["evidence-official-a248990b44d481814f47","evidence-official-849b751ed4bf88622749","evidence-official-b8df803e8f55324a4eb5","evidence-official-8717ffb993cc8f7eddec"],"source_locator":"Evo 2 paper: architecture, training, and data, Figure 1 and Discussion; official checkpoints table and evo2_7b licence metadata"}],"diagram":{"title":"Evo 2 workflow","steps":["DNA bases","StripedHyena 2","Autoregressive outputs","Sequence scoring or generation"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-a248990b44d481814f47","evidence-official-849b751ed4bf88622749","evidence-official-b8df803e8f55324a4eb5","evidence-official-8717ffb993cc8f7eddec"],"source_locator":"Evo 2 paper: architecture, training, and data, Figure 1 and Discussion; official checkpoints table and evo2_7b licence metadata"},"coverage":"limited","gaps":["Training cutoff: OpenGenome2 combines multiple nucleotide collections. The inspected paper and released checkpoint documentation do not provide one latest-deposition date that covers every component."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}},"description":"Genome sequence modelling family","facets":{"areas":["genomics"]},"id":"discovery-model-evo-2","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-geneb"}],"name":"Evo 2","source_ids":["src-discovery-arcinstitute-evo2"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-dart-eval"],"entity_level":"method","reported_name":"FIMO","version":null,"historical_missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"FIMO scans sequences for occurrences of supplied sequence motifs.","summary_source_ids":["evidence-official-f0b17e30e69338b7762f","evidence-official-94c61d963d28fa3a8993"],"summary_source_locator":"Official FIMO page: motif input, sequence input, p-value threshold and strand options","sections":[{"title":"How it works","body":"FIMO scans sequences for occurrences of supplied sequence motifs. Position-specific motif scanning with statistical match thresholds. The documented inputs are DNA, RNA or protein sequences and motif matrices in the supported alphabet/formats. The output consists of motif matches at sequence positions subject to the configured statistical threshold.","source_ids":["evidence-official-f0b17e30e69338b7762f","evidence-official-94c61d963d28fa3a8993"],"source_locator":"Official FIMO page: motif input, sequence input, p-value threshold and strand options"},{"title":"Versions and reproducibility","body":"MEME Suite FIMO; site snapshot pinned by retrieval time and hash. Input sequences and motif lengths, not a fixed language-model context.","source_ids":["evidence-official-f0b17e30e69338b7762f","evidence-official-94c61d963d28fa3a8993"],"source_locator":"Official FIMO page: motif input, sequence input, p-value threshold and strand options"}],"facts":[{"label":"Model type","value":"Statistical motif scanning procedure","status":"source_checked","source_ids":["evidence-official-f0b17e30e69338b7762f","evidence-official-94c61d963d28fa3a8993"],"source_locator":"Official FIMO page: motif input, sequence input, p-value threshold and strand options"},{"label":"Architecture","value":"Position-specific motif scanning with statistical match thresholds.","status":"source_checked","source_ids":["evidence-official-f0b17e30e69338b7762f","evidence-official-94c61d963d28fa3a8993"],"source_locator":"Official FIMO page: motif input, sequence input, p-value threshold and strand options"},{"label":"Inputs","value":"DNA, RNA or protein sequences and motif matrices in the supported alphabet/formats.","status":"source_checked","source_ids":["evidence-official-f0b17e30e69338b7762f","evidence-official-94c61d963d28fa3a8993"],"source_locator":"Official FIMO page: motif input, sequence input, p-value threshold and strand options"},{"label":"Outputs","value":"Motif matches at sequence positions subject to the configured statistical threshold.","status":"source_checked","source_ids":["evidence-official-f0b17e30e69338b7762f","evidence-official-94c61d963d28fa3a8993"],"source_locator":"Official FIMO page: motif input, sequence input, p-value threshold and strand options"},{"label":"Parameters","value":"Motif probabilities and scan settings; not a neural parameter count.","status":"source_checked","source_ids":["evidence-official-f0b17e30e69338b7762f","evidence-official-94c61d963d28fa3a8993"],"source_locator":"Official FIMO page: motif input, sequence input, p-value threshold and strand options"},{"label":"Known versions","value":"MEME Suite FIMO; site snapshot pinned by retrieval time and hash.","status":"source_checked","source_ids":["evidence-official-f0b17e30e69338b7762f","evidence-official-94c61d963d28fa3a8993"],"source_locator":"Official FIMO page: motif input, sequence input, p-value threshold and strand options"},{"label":"Training data","value":"Motifs are supplied by the user or a selected database. FIMO is the scanning procedure, not the motif-training method.","status":"source_checked","source_ids":["evidence-official-f0b17e30e69338b7762f","evidence-official-94c61d963d28fa3a8993"],"source_locator":"Official FIMO page: motif input, sequence input, p-value threshold and strand options"},{"label":"Training cutoff","value":"Inapplicable to neural pretraining; reference-database and input-data dates must be recorded for each run.","status":"inapplicable","source_ids":["evidence-official-f0b17e30e69338b7762f","evidence-official-94c61d963d28fa3a8993"],"source_locator":"Official FIMO page: motif input, sequence input, p-value threshold and strand options"},{"label":"Context limits","value":"Input sequences and motif lengths, not a fixed language-model context.","status":"source_checked","source_ids":["evidence-official-f0b17e30e69338b7762f","evidence-official-94c61d963d28fa3a8993"],"source_locator":"Official FIMO page: motif input, sequence input, p-value threshold and strand options"},{"label":"Weights licence","value":"Inapplicable: supplied motif matrices replace neural weights.","status":"inapplicable","source_ids":["evidence-official-f0b17e30e69338b7762f","evidence-official-94c61d963d28fa3a8993"],"source_locator":"Official FIMO page: motif input, sequence input, p-value threshold and strand options"},{"label":"Access","value":"Official project documentation and implementation: https://meme-suite.org/meme/tools/fimo","status":"source_checked","source_ids":["evidence-official-f0b17e30e69338b7762f","evidence-official-94c61d963d28fa3a8993"],"source_locator":"Official FIMO page: motif input, sequence input, p-value threshold and strand options"},{"label":"Code licence","value":"MEME Suite custom terms permit educational, research and non-profit use with notices retained; commercial use requires contacting UC San Diego for terms.","status":"source_checked","source_ids":["evidence-official-94c61d963d28fa3a8993"],"source_locator":"MEME Suite copyright notice: permission grant and commercial-use paragraph"}],"strengths":[{"text":"Provides a simple motif-based reference whose supplied motif and background assumptions can be reported explicitly.","source_ids":["evidence-official-f0b17e30e69338b7762f","evidence-official-94c61d963d28fa3a8993"],"source_locator":"Official FIMO page: motif input, sequence input, p-value threshold and strand options"}],"limitations":[{"text":"Motif matches do not establish molecular binding or regulation by themselves. Alphabet, strand and threshold choices affect the scan.","source_ids":["evidence-official-f0b17e30e69338b7762f","evidence-official-94c61d963d28fa3a8993"],"source_locator":"Official FIMO page: motif input, sequence input, p-value threshold and strand options"}],"diagram":{"title":"FIMO workflow","steps":["Motif matrices and sequence","Scan motif positions","Apply statistical threshold","Reported motif occurrences"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-f0b17e30e69338b7762f","evidence-official-94c61d963d28fa3a8993"],"source_locator":"Official FIMO page: motif input, sequence input, p-value threshold and strand options"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}},"description":"Sequence motif scanning","facets":{"areas":["genomics"]},"id":"discovery-model-fimo","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-dart-eval"}],"name":"FIMO","source_ids":["src-discovery-meme"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-perturbench"],"entity_level":"family","reported_name":"GEARS","version":null,"historical_missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"GEARS predicts transcriptional responses to single- and multi-gene perturbations from single-cell perturbation screens.","summary_source_ids":["evidence-official-c037e3419c04936262a0","evidence-official-b5e81b2c497214b18f8a","evidence-official-f464cd407a6a9d031720"],"summary_source_locator":"GEARS paper Methods: Overview, Gene coexpression graph encoder, GO graph, Cross-gene effects and gene-specific decoder; gears/model.py: GEARS_Model","sections":[{"title":"How it works","body":"GEARS predicts transcriptional responses to single- and multi-gene perturbations from single-cell perturbation screens. Two graph encoders represent gene coexpression and Gene Ontology perturbation similarity. Perturbation embeddings are composed with gene embeddings, then a cross-gene network and gene-specific decoders predict expression changes. The documented inputs are single-cell expression data, perturbation labels and the graph resources used by the configured model. The output consists of predicted gene-expression responses and genetic-interaction analyses.","source_ids":["evidence-official-c037e3419c04936262a0","evidence-official-b5e81b2c497214b18f8a","evidence-official-f464cd407a6a9d031720"],"source_locator":"GEARS paper Methods: Overview, Gene coexpression graph encoder, GO graph, Cross-gene effects and gene-specific decoder; gears/model.py: GEARS_Model"},{"title":"Versions and reproducibility","body":"The README describes v0.1.1 updates; a specific trained checkpoint must be recorded separately. A gene-expression vector and perturbation set over the configured gene inventory; no fixed nucleotide or amino-acid token window.","source_ids":["evidence-official-c037e3419c04936262a0","evidence-official-b5e81b2c497214b18f8a","evidence-official-f464cd407a6a9d031720"],"source_locator":"GEARS paper Methods: Overview, Gene coexpression graph encoder, GO graph, Cross-gene effects and gene-specific decoder; gears/model.py: GEARS_Model"}],"facts":[{"label":"Model type","value":"Graph-based perturbation-response predictor","status":"source_checked","source_ids":["evidence-official-c037e3419c04936262a0","evidence-official-b5e81b2c497214b18f8a","evidence-official-f464cd407a6a9d031720"],"source_locator":"GEARS paper Methods: Overview, Gene coexpression graph encoder, GO graph, Cross-gene effects and gene-specific decoder; gears/model.py: GEARS_Model"},{"label":"Architecture","value":"Two graph encoders represent gene coexpression and Gene Ontology perturbation similarity. Perturbation embeddings are composed with gene embeddings, then a cross-gene network and gene-specific decoders predict expression changes.","status":"source_checked","source_ids":["evidence-official-c037e3419c04936262a0","evidence-official-b5e81b2c497214b18f8a","evidence-official-f464cd407a6a9d031720"],"source_locator":"GEARS paper Methods: Overview, Gene coexpression graph encoder, GO graph, Cross-gene effects and gene-specific decoder; gears/model.py: GEARS_Model"},{"label":"Inputs","value":"Single-cell expression data, perturbation labels and the graph resources used by the configured model.","status":"source_checked","source_ids":["evidence-official-c037e3419c04936262a0","evidence-official-b5e81b2c497214b18f8a","evidence-official-f464cd407a6a9d031720"],"source_locator":"GEARS paper Methods: Overview, Gene coexpression graph encoder, GO graph, Cross-gene effects and gene-specific decoder; gears/model.py: GEARS_Model"},{"label":"Outputs","value":"Predicted gene-expression responses and genetic-interaction analyses.","status":"source_checked","source_ids":["evidence-official-c037e3419c04936262a0","evidence-official-b5e81b2c497214b18f8a","evidence-official-f464cd407a6a9d031720"],"source_locator":"GEARS paper Methods: Overview, Gene coexpression graph encoder, GO graph, Cross-gene effects and gene-specific decoder; gears/model.py: GEARS_Model"},{"label":"Parameters","value":"Configuration-dependent: gene and perturbation embedding tables grow with the selected gene/perturbation inventory, alongside graph and decoder parameters.","status":"source_checked","source_ids":["evidence-official-c037e3419c04936262a0","evidence-official-b5e81b2c497214b18f8a","evidence-official-f464cd407a6a9d031720"],"source_locator":"GEARS paper Methods: Overview, Gene coexpression graph encoder, GO graph, Cross-gene effects and gene-specific decoder; gears/model.py: GEARS_Model"},{"label":"Known versions","value":"The README describes v0.1.1 updates; a specific trained checkpoint must be recorded separately.","status":"source_checked","source_ids":["evidence-official-c037e3419c04936262a0","evidence-official-b5e81b2c497214b18f8a","evidence-official-f464cd407a6a9d031720"],"source_locator":"GEARS paper Methods: Overview, Gene coexpression graph encoder, GO graph, Cross-gene effects and gene-specific decoder; gears/model.py: GEARS_Model"},{"label":"Training data","value":"Fitted to the selected perturbation screen. Examples include Norman, Adamson and Dixit; the repository also lists Replogle RPE1/K562 loaders.","status":"source_checked","source_ids":["evidence-official-c037e3419c04936262a0","evidence-official-b5e81b2c497214b18f8a","evidence-official-f464cd407a6a9d031720"],"source_locator":"GEARS paper Methods: Overview, Gene coexpression graph encoder, GO graph, Cross-gene effects and gene-specific decoder; gears/model.py: GEARS_Model"},{"label":"Training cutoff","value":"Inapplicable as a universal pretrained-model cutoff: GEARS fits the provided perturbation training set and builds its coexpression graph from that set.","status":"inapplicable","source_ids":["evidence-official-c037e3419c04936262a0","evidence-official-b5e81b2c497214b18f8a","evidence-official-f464cd407a6a9d031720"],"source_locator":"GEARS paper Methods: Overview, Gene coexpression graph encoder, GO graph, Cross-gene effects and gene-specific decoder; gears/model.py: GEARS_Model"},{"label":"Context limits","value":"A gene-expression vector and perturbation set over the configured gene inventory; no fixed nucleotide or amino-acid token window.","status":"source_checked","source_ids":["evidence-official-c037e3419c04936262a0","evidence-official-b5e81b2c497214b18f8a","evidence-official-f464cd407a6a9d031720"],"source_locator":"GEARS paper Methods: Overview, Gene coexpression graph encoder, GO graph, Cross-gene effects and gene-specific decoder; gears/model.py: GEARS_Model"},{"label":"Weights licence","value":"Separate checkpoint-distribution terms are not stated in the inspected release documentation and licence material. The source-code licence alone is not recorded as an explicit weight grant.","status":"unreported","source_ids":["evidence-official-c037e3419c04936262a0","evidence-official-b5e81b2c497214b18f8a","evidence-official-f464cd407a6a9d031720","evidence-official-f06a8695b2915f86a45a"],"source_locator":"GEARS paper Methods: Overview, Gene coexpression graph encoder, GO graph, Cross-gene effects and gene-specific decoder; gears/model.py: GEARS_Model; LICENSE: licence text"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/snap-stanford/GEARS","status":"source_checked","source_ids":["evidence-official-c037e3419c04936262a0","evidence-official-b5e81b2c497214b18f8a","evidence-official-f464cd407a6a9d031720"],"source_locator":"GEARS paper Methods: Overview, Gene coexpression graph encoder, GO graph, Cross-gene effects and gene-specific decoder; gears/model.py: GEARS_Model"},{"label":"Code licence","value":"MIT","status":"source_checked","source_ids":["evidence-official-f06a8695b2915f86a45a"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"Provides documented dataset/split handling and training interfaces for single and combinatorial perturbations.","source_ids":["evidence-official-c037e3419c04936262a0","evidence-official-b5e81b2c497214b18f8a","evidence-official-f464cd407a6a9d031720"],"source_locator":"GEARS paper Methods: Overview, Gene coexpression graph encoder, GO graph, Cross-gene effects and gene-specific decoder; gears/model.py: GEARS_Model"}],"limitations":[{"text":"The authors explicitly warn against cross-cell-type transfer, bulk-RNA assumptions and predicting combinations after training only on single perturbations.","source_ids":["evidence-official-c037e3419c04936262a0","evidence-official-b5e81b2c497214b18f8a","evidence-official-f464cd407a6a9d031720"],"source_locator":"GEARS paper Methods: Overview, Gene coexpression graph encoder, GO graph, Cross-gene effects and gene-specific decoder; gears/model.py: GEARS_Model"}],"diagram":{"title":"GEARS workflow","steps":["Perturbation screen","Gene and perturbation graph representations","GEARS prediction","Expression response"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-c037e3419c04936262a0","evidence-official-b5e81b2c497214b18f8a","evidence-official-f464cd407a6a9d031720"],"source_locator":"GEARS paper Methods: Overview, Gene coexpression graph encoder, GO graph, Cross-gene effects and gene-specific decoder; gears/model.py: GEARS_Model"},"coverage":"limited","gaps":["Weights licence: Separate checkpoint-distribution terms are not stated in the inspected release documentation and licence material. The source-code licence alone is not recorded as an explicit weight grant."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}},"description":"Genetic perturbation response prediction","facets":{"areas":["single-cell"]},"id":"discovery-model-gears","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-perturbench"}],"name":"GEARS","source_ids":["src-discovery-snap-stanford-gears"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":[],"entity_level":"family","reported_name":"Genie 3","version":null,"historical_missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"Genie 3 generates protein designs through all-atom equivariant diffusion.","summary_source_ids":["evidence-official-082d60e1af5af7ed12ca","evidence-official-a07254268b1d1de1f7f8"],"summary_source_locator":"README.md: overview, Download model weights and training data, Training and Codebase Architecture; author-linked yeqinglin/genie3 licence metadata; README.md: Training / Dataset manifest table","sections":[{"title":"How it works","body":"Genie 3 generates protein designs through all-atom equivariant diffusion. All-atom SE(3)-equivariant diffusion model, with separate generation and evaluation workflows. The documented inputs are unconditional design specification, motif constraints or binder-design target context. The output consists of sampled protein designs and outputs from the selected downstream evaluation workflow.","source_ids":["evidence-official-082d60e1af5af7ed12ca","evidence-official-a07254268b1d1de1f7f8"],"source_locator":"README.md: overview, Download model weights and training data, Training and Codebase Architecture; author-linked yeqinglin/genie3 licence metadata; README.md: Training / Dataset manifest table"},{"title":"Versions and reproducibility","body":"Genie 3; repository includes compatibility guidance for Genie 2. The applicable input limits require configuration-specific checking.","source_ids":["evidence-official-082d60e1af5af7ed12ca","evidence-official-a07254268b1d1de1f7f8"],"source_locator":"README.md: overview, Download model weights and training data, Training and Codebase Architecture; author-linked yeqinglin/genie3 licence metadata; README.md: Training / Dataset manifest table"}],"facts":[{"label":"Model type","value":"Diffusion-based protein backbone generator","status":"source_checked","source_ids":["evidence-official-082d60e1af5af7ed12ca","evidence-official-a07254268b1d1de1f7f8"],"source_locator":"README.md: overview, Download model weights and training data, Training and Codebase Architecture; author-linked yeqinglin/genie3 licence metadata; README.md: Training / Dataset manifest table"},{"label":"Architecture","value":"All-atom SE(3)-equivariant diffusion model, with separate generation and evaluation workflows.","status":"source_checked","source_ids":["evidence-official-082d60e1af5af7ed12ca","evidence-official-a07254268b1d1de1f7f8"],"source_locator":"README.md: overview, Download model weights and training data, Training and Codebase Architecture; author-linked yeqinglin/genie3 licence metadata; README.md: Training / Dataset manifest table"},{"label":"Inputs","value":"Unconditional design specification, motif constraints or binder-design target context.","status":"source_checked","source_ids":["evidence-official-082d60e1af5af7ed12ca","evidence-official-a07254268b1d1de1f7f8"],"source_locator":"README.md: overview, Download model weights and training data, Training and Codebase Architecture; author-linked yeqinglin/genie3 licence metadata; README.md: Training / Dataset manifest table"},{"label":"Outputs","value":"Sampled protein designs and outputs from the selected downstream evaluation workflow.","status":"source_checked","source_ids":["evidence-official-082d60e1af5af7ed12ca","evidence-official-a07254268b1d1de1f7f8"],"source_locator":"README.md: overview, Download model weights and training data, Training and Codebase Architecture; author-linked yeqinglin/genie3 licence metadata; README.md: Training / Dataset manifest table"},{"label":"Parameters","value":"The inspected release README and model card do not state a complete parameter total for the released all-atom diffusion checkpoint.","status":"unreported","source_ids":["evidence-official-082d60e1af5af7ed12ca","evidence-official-a07254268b1d1de1f7f8"],"source_locator":"README.md: overview, Download model weights and training data, Training and Codebase Architecture; author-linked yeqinglin/genie3 licence metadata; README.md: Training / Dataset manifest table"},{"label":"Known versions","value":"Genie 3; repository includes compatibility guidance for Genie 2.","status":"source_checked","source_ids":["evidence-official-082d60e1af5af7ed12ca","evidence-official-a07254268b1d1de1f7f8"],"source_locator":"README.md: overview, Download model weights and training data, Training and Codebase Architecture; author-linked yeqinglin/genie3 licence metadata; README.md: Training / Dataset manifest table"},{"label":"Training data","value":"The released training manifests cover AlphaFoldDB representatives of at most 512 residues with pLDDT at least 70, and PiNDER 2024-02. These filters describe training components, not inference limits.","status":"source_checked","source_ids":["evidence-official-082d60e1af5af7ed12ca","evidence-official-a07254268b1d1de1f7f8"],"source_locator":"README.md: overview, Download model weights and training data, Training and Codebase Architecture; author-linked yeqinglin/genie3 licence metadata; README.md: Training / Dataset manifest table"},{"label":"Training cutoff","value":"The documented training inputs include PiNDER 2024-02 and filtered AlphaFoldDB representatives. This identifies a PiNDER release, not a universal latest-deposition cutoff for all training data.","status":"source_checked","source_ids":["evidence-official-082d60e1af5af7ed12ca","evidence-official-a07254268b1d1de1f7f8"],"source_locator":"README.md: overview, Download model weights and training data, Training and Codebase Architecture; author-linked yeqinglin/genie3 licence metadata; README.md: Training / Dataset manifest table"},{"label":"Context limits","value":"The README permits configured design-length ranges and distinguishes sampling settings above and below 300 residues; it does not state one validated maximum for all monomer, motif and binder tasks.","status":"unreported","source_ids":["evidence-official-082d60e1af5af7ed12ca","evidence-official-a07254268b1d1de1f7f8"],"source_locator":"README.md: overview, Download model weights and training data, Training and Codebase Architecture; author-linked yeqinglin/genie3 licence metadata; README.md: Training / Dataset manifest table"},{"label":"Weights licence","value":"Apache-2.0 declared in the author-linked yeqinglin/genie3 model card.","status":"source_checked","source_ids":["evidence-official-082d60e1af5af7ed12ca","evidence-official-a07254268b1d1de1f7f8"],"source_locator":"README.md: overview, Download model weights and training data, Training and Codebase Architecture; author-linked yeqinglin/genie3 licence metadata; README.md: Training / Dataset manifest table"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/aqlaboratory/genie3","status":"source_checked","source_ids":["evidence-official-082d60e1af5af7ed12ca","evidence-official-a07254268b1d1de1f7f8"],"source_locator":"README.md: overview, Download model weights and training data, Training and Codebase Architecture; author-linked yeqinglin/genie3 licence metadata; README.md: Training / Dataset manifest table"},{"label":"Code licence","value":"Apache-2.0","status":"source_checked","source_ids":["evidence-official-bbfa62298e3bb7c5494d"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"One documented interface supports unconditional generation, motif scaffolding and binder design.","source_ids":["evidence-official-082d60e1af5af7ed12ca","evidence-official-a07254268b1d1de1f7f8"],"source_locator":"README.md: overview, Download model weights and training data, Training and Codebase Architecture; author-linked yeqinglin/genie3 licence metadata; README.md: Training / Dataset manifest table"}],"limitations":[{"text":"Genie 3 protein design is unrelated to GENIE3 gene-regulatory-network inference. A computationally generated design needs separate experimental validation.","source_ids":["evidence-official-082d60e1af5af7ed12ca","evidence-official-a07254268b1d1de1f7f8"],"source_locator":"README.md: overview, Download model weights and training data, Training and Codebase Architecture; author-linked yeqinglin/genie3 licence metadata; README.md: Training / Dataset manifest table"}],"diagram":{"title":"Genie 3 workflow","steps":["Design constraints","Equivariant diffusion","All-atom design","Configured evaluation"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-082d60e1af5af7ed12ca","evidence-official-a07254268b1d1de1f7f8"],"source_locator":"README.md: overview, Download model weights and training data, Training and Codebase Architecture; author-linked yeqinglin/genie3 licence metadata; README.md: Training / Dataset manifest table"},"coverage":"limited","gaps":["Parameters: The inspected release README and model card do not state a complete parameter total for the released all-atom diffusion checkpoint.","Context limits: The README permits configured design-length ranges and distinguishes sampling settings above and below 300 residues; it does not state one validated maximum for all monomer, motif and binder tasks."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}},"description":"Equivariant all-atom protein design","facets":{"areas":["protein-structure"]},"id":"discovery-model-genie-3","kind":"model","links":[],"name":"Genie 3","source_ids":["src-discovery-aqlaboratory-genie3"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-beeline"],"entity_level":"method","reported_name":"GENIE3","version":null,"historical_missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"GENIE3 infers candidate gene-regulatory links from expression measurements using ensembles of trees.","summary_source_ids":["evidence-official-f078ec10fa45b5d77ad7","evidence-official-ad7f5c1eb802d9895413","evidence-official-099d109417c7fae96e96"],"summary_source_locator":"README.md: GENIE3 definition; GRNBoost README Introduction describes the GENIE3 target-wise regression blueprint","sections":[{"title":"How it works","body":"GENIE3 infers candidate gene-regulatory links from expression measurements using ensembles of trees. Tree-ensemble regression decomposes network inference into prediction of each target gene from candidate regulators. The documented inputs are gene-expression matrix and the selected candidate-regulator set. The output consists of ranked candidate regulator-to-target links.","source_ids":["evidence-official-f078ec10fa45b5d77ad7","evidence-official-ad7f5c1eb802d9895413","evidence-official-099d109417c7fae96e96"],"source_locator":"README.md: GENIE3 definition; GRNBoost README Introduction describes the GENIE3 target-wise regression blueprint"},{"title":"Versions and reproducibility","body":"GENIE3 implementation and Bioconductor package; preserve the configured tree method and settings. Genes and observations in the input matrix; no token context.","source_ids":["evidence-official-f078ec10fa45b5d77ad7","evidence-official-ad7f5c1eb802d9895413","evidence-official-099d109417c7fae96e96"],"source_locator":"README.md: GENIE3 definition; GRNBoost README Introduction describes the GENIE3 target-wise regression blueprint"}],"facts":[{"label":"Model type","value":"Tree-ensemble gene-regulatory-network inference","status":"source_checked","source_ids":["evidence-official-f078ec10fa45b5d77ad7","evidence-official-ad7f5c1eb802d9895413","evidence-official-099d109417c7fae96e96"],"source_locator":"README.md: GENIE3 definition; GRNBoost README Introduction describes the GENIE3 target-wise regression blueprint"},{"label":"Architecture","value":"Tree-ensemble regression decomposes network inference into prediction of each target gene from candidate regulators.","status":"source_checked","source_ids":["evidence-official-f078ec10fa45b5d77ad7","evidence-official-ad7f5c1eb802d9895413","evidence-official-099d109417c7fae96e96"],"source_locator":"README.md: GENIE3 definition; GRNBoost README Introduction describes the GENIE3 target-wise regression blueprint"},{"label":"Inputs","value":"Gene-expression matrix and the selected candidate-regulator set.","status":"source_checked","source_ids":["evidence-official-f078ec10fa45b5d77ad7","evidence-official-ad7f5c1eb802d9895413","evidence-official-099d109417c7fae96e96"],"source_locator":"README.md: GENIE3 definition; GRNBoost README Introduction describes the GENIE3 target-wise regression blueprint"},{"label":"Outputs","value":"Ranked candidate regulator-to-target links.","status":"source_checked","source_ids":["evidence-official-f078ec10fa45b5d77ad7","evidence-official-ad7f5c1eb802d9895413","evidence-official-099d109417c7fae96e96"],"source_locator":"README.md: GENIE3 definition; GRNBoost README Introduction describes the GENIE3 target-wise regression blueprint"},{"label":"Parameters","value":"Fitted tree ensembles; count depends on genes, candidate regulators and tree settings.","status":"source_checked","source_ids":["evidence-official-f078ec10fa45b5d77ad7","evidence-official-ad7f5c1eb802d9895413","evidence-official-099d109417c7fae96e96"],"source_locator":"README.md: GENIE3 definition; GRNBoost README Introduction describes the GENIE3 target-wise regression blueprint"},{"label":"Known versions","value":"GENIE3 implementation and Bioconductor package; preserve the configured tree method and settings.","status":"source_checked","source_ids":["evidence-official-f078ec10fa45b5d77ad7","evidence-official-ad7f5c1eb802d9895413","evidence-official-099d109417c7fae96e96"],"source_locator":"README.md: GENIE3 definition; GRNBoost README Introduction describes the GENIE3 target-wise regression blueprint"},{"label":"Training data","value":"Fitted to the supplied gene-expression dataset, not a universal pretrained corpus.","status":"source_checked","source_ids":["evidence-official-f078ec10fa45b5d77ad7","evidence-official-ad7f5c1eb802d9895413","evidence-official-099d109417c7fae96e96"],"source_locator":"README.md: GENIE3 definition; GRNBoost README Introduction describes the GENIE3 target-wise regression blueprint"},{"label":"Training cutoff","value":"Inapplicable to neural pretraining; reference-database and input-data dates must be recorded for each run.","status":"inapplicable","source_ids":["evidence-official-f078ec10fa45b5d77ad7","evidence-official-ad7f5c1eb802d9895413","evidence-official-099d109417c7fae96e96"],"source_locator":"README.md: GENIE3 definition; GRNBoost README Introduction describes the GENIE3 target-wise regression blueprint"},{"label":"Context limits","value":"Genes and observations in the input matrix; no token context.","status":"source_checked","source_ids":["evidence-official-f078ec10fa45b5d77ad7","evidence-official-ad7f5c1eb802d9895413","evidence-official-099d109417c7fae96e96"],"source_locator":"README.md: GENIE3 definition; GRNBoost README Introduction describes the GENIE3 target-wise regression blueprint"},{"label":"Weights licence","value":"No universal pretrained checkpoint; each fitted network is dataset-specific.","status":"source_checked","source_ids":["evidence-official-f078ec10fa45b5d77ad7","evidence-official-ad7f5c1eb802d9895413","evidence-official-099d109417c7fae96e96"],"source_locator":"README.md: GENIE3 definition; GRNBoost README Introduction describes the GENIE3 target-wise regression blueprint"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/aertslab/GENIE3","status":"source_checked","source_ids":["evidence-official-f078ec10fa45b5d77ad7","evidence-official-ad7f5c1eb802d9895413","evidence-official-099d109417c7fae96e96"],"source_locator":"README.md: GENIE3 definition; GRNBoost README Introduction describes the GENIE3 target-wise regression blueprint"},{"label":"Code licence","value":"GPL-2.0-or-later, as declared in the R package DESCRIPTION.","status":"source_checked","source_ids":["evidence-official-ad7f5c1eb802d9895413"],"source_locator":"DESCRIPTION: License"}],"strengths":[{"text":"A conventional machine-learning comparator whose target-wise regression and regulator importance are explicit.","source_ids":["evidence-official-f078ec10fa45b5d77ad7","evidence-official-ad7f5c1eb802d9895413","evidence-official-099d109417c7fae96e96"],"source_locator":"README.md: GENIE3 definition; GRNBoost README Introduction describes the GENIE3 target-wise regression blueprint"}],"limitations":[{"text":"Predictive association in expression data does not by itself establish causal regulation. This GENIE3 algorithm is unrelated to Genie 3 protein design.","source_ids":["evidence-official-f078ec10fa45b5d77ad7","evidence-official-ad7f5c1eb802d9895413","evidence-official-099d109417c7fae96e96"],"source_locator":"README.md: GENIE3 definition; GRNBoost README Introduction describes the GENIE3 target-wise regression blueprint"}],"diagram":{"title":"GENIE3 workflow","steps":["Expression matrix","Target-wise tree ensembles","Regulator importance","Candidate network links"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-f078ec10fa45b5d77ad7","evidence-official-ad7f5c1eb802d9895413","evidence-official-099d109417c7fae96e96"],"source_locator":"README.md: GENIE3 definition; GRNBoost README Introduction describes the GENIE3 target-wise regression blueprint"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}},"description":"Tree-ensemble gene regulatory network inference","facets":{"areas":["biological-networks"]},"id":"discovery-model-genie3","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-beeline"}],"name":"GENIE3","source_ids":["src-discovery-aertslab-genie3"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-glycanml"],"entity_level":"family","reported_name":"GlycanGT","version":null,"historical_missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"GlycanGT learns glycan representations by treating monosaccharides and linkages as graph tokens.","summary_source_ids":["evidence-official-10a6550ca70fdeaf33cb","evidence-official-790ab6283e5bb91474ed","evidence-official-dacb012b22c0a7de4880","evidence-official-87c70bddba97a72a5e5d","evidence-official-ba12a2d759aba6800ae6"],"summary_source_locator":"GlycanGT paper Sections 2.1–2.4,2.9 and 4; official model-card licence metadata; primary paper Sections 2.1 and 3.1 versus README Model architecture / Training details; model/config_tokengt.py: size presets; publisher supplementary Table S2, Total Parameters row","sections":[{"title":"How it works","body":"GlycanGT turns each monosaccharide and glycosidic linkage into a token. Content embeddings, orthogonal node identifiers and token-type embeddings preserve graph structure before transformer attention. A graph token provides the whole-glycan representation; masked-token heads can suggest missing components, while downstream classifiers are fitted separately.","source_ids":["evidence-official-10a6550ca70fdeaf33cb","evidence-official-790ab6283e5bb91474ed","evidence-official-dacb012b22c0a7de4880","evidence-official-87c70bddba97a72a5e5d","evidence-official-ba12a2d759aba6800ae6"],"source_locator":"GlycanGT paper Sections 2.1–2.4,2.9 and 4; official model-card licence metadata; primary paper Sections 2.1 and 3.1 versus README Model architecture / Training details; model/config_tokengt.py: size presets; publisher supplementary Table S2, Total Parameters row"},{"title":"Versions and reproducibility","body":"ss, small, medium and large scales; the documented downstream work uses the large model with 35% masking. The applicable input limits require configuration-specific checking.","source_ids":["evidence-official-10a6550ca70fdeaf33cb","evidence-official-790ab6283e5bb91474ed","evidence-official-dacb012b22c0a7de4880","evidence-official-87c70bddba97a72a5e5d","evidence-official-ba12a2d759aba6800ae6"],"source_locator":"GlycanGT paper Sections 2.1–2.4,2.9 and 4; official model-card licence metadata; primary paper Sections 2.1 and 3.1 versus README Model architecture / Training details; model/config_tokengt.py: size presets; publisher supplementary Table S2, Total Parameters row"}],"facts":[{"label":"Model type","value":"Glycan graph transformer","status":"source_checked","source_ids":["evidence-official-10a6550ca70fdeaf33cb","evidence-official-790ab6283e5bb91474ed","evidence-official-dacb012b22c0a7de4880","evidence-official-87c70bddba97a72a5e5d","evidence-official-ba12a2d759aba6800ae6"],"source_locator":"GlycanGT paper Sections 2.1–2.4,2.9 and 4; official model-card licence metadata; primary paper Sections 2.1 and 3.1 versus README Model architecture / Training details; model/config_tokengt.py: size presets; publisher supplementary Table S2, Total Parameters row"},{"label":"Architecture","value":"TokenGT-derived graph transformer with content features, orthogonal node identifiers, node/edge type embeddings and a graph token.","status":"source_checked","source_ids":["evidence-official-10a6550ca70fdeaf33cb","evidence-official-790ab6283e5bb91474ed","evidence-official-dacb012b22c0a7de4880","evidence-official-87c70bddba97a72a5e5d","evidence-official-ba12a2d759aba6800ae6"],"source_locator":"GlycanGT paper Sections 2.1–2.4,2.9 and 4; official model-card licence metadata; primary paper Sections 2.1 and 3.1 versus README Model architecture / Training details; model/config_tokengt.py: size presets; publisher supplementary Table S2, Total Parameters row"},{"label":"Inputs","value":"Glycan graphs with monosaccharide nodes and glycosidic-linkage edges.","status":"source_checked","source_ids":["evidence-official-10a6550ca70fdeaf33cb","evidence-official-790ab6283e5bb91474ed","evidence-official-dacb012b22c0a7de4880","evidence-official-87c70bddba97a72a5e5d","evidence-official-ba12a2d759aba6800ae6"],"source_locator":"GlycanGT paper Sections 2.1–2.4,2.9 and 4; official model-card licence metadata; primary paper Sections 2.1 and 3.1 versus README Model architecture / Training details; model/config_tokengt.py: size presets; publisher supplementary Table S2, Total Parameters row"},{"label":"Outputs","value":"Graph embeddings and masked-component predictions for incomplete glycans.","status":"source_checked","source_ids":["evidence-official-10a6550ca70fdeaf33cb","evidence-official-790ab6283e5bb91474ed","evidence-official-dacb012b22c0a7de4880","evidence-official-87c70bddba97a72a5e5d","evidence-official-ba12a2d759aba6800ae6"],"source_locator":"GlycanGT paper Sections 2.1–2.4,2.9 and 4; official model-card licence metadata; primary paper Sections 2.1 and 3.1 versus README Model architecture / Training details; model/config_tokengt.py: size presets; publisher supplementary Table S2, Total Parameters row"},{"label":"Parameters","value":"ss: 1,922,002; small: 6,208,722; medium: 29,731,026; large: 91,820,242 total parameters, as printed in supplementary Table S2.","status":"source_checked","source_ids":["evidence-official-10a6550ca70fdeaf33cb","evidence-official-790ab6283e5bb91474ed","evidence-official-dacb012b22c0a7de4880","evidence-official-87c70bddba97a72a5e5d","evidence-official-ba12a2d759aba6800ae6"],"source_locator":"GlycanGT paper Sections 2.1–2.4,2.9 and 4; official model-card licence metadata; primary paper Sections 2.1 and 3.1 versus README Model architecture / Training details; model/config_tokengt.py: size presets; publisher supplementary Table S2, Total Parameters row"},{"label":"Known versions","value":"ss, small, medium and large scales; the documented downstream work uses the large model with 35% masking.","status":"source_checked","source_ids":["evidence-official-10a6550ca70fdeaf33cb","evidence-official-790ab6283e5bb91474ed","evidence-official-dacb012b22c0a7de4880","evidence-official-87c70bddba97a72a5e5d","evidence-official-ba12a2d759aba6800ae6"],"source_locator":"GlycanGT paper Sections 2.1–2.4,2.9 and 4; official model-card licence metadata; primary paper Sections 2.1 and 3.1 versus README Model architecture / Training details; model/config_tokengt.py: size presets; publisher supplementary Table S2, Total Parameters row"},{"label":"Training data","value":"The primary paper reports 83,739 curated glycans after ambiguity and downstream-overlap exclusions; the repository README reports 83,740. Both source values are preserved pending an author correction.","status":"source_checked","source_ids":["evidence-official-10a6550ca70fdeaf33cb","evidence-official-790ab6283e5bb91474ed","evidence-official-dacb012b22c0a7de4880","evidence-official-87c70bddba97a72a5e5d","evidence-official-ba12a2d759aba6800ae6"],"source_locator":"GlycanGT paper Sections 2.1–2.4,2.9 and 4; official model-card licence metadata; primary paper Sections 2.1 and 3.1 versus README Model architecture / Training details; model/config_tokengt.py: size presets; publisher supplementary Table S2, Total Parameters row"},{"label":"Training cutoff","value":"GlyCosmos/GlyTouCan source retrieval: 17 June 2025. The paper describes removal of downstream-benchmark overlaps from pretraining.","status":"source_checked","source_ids":["evidence-official-10a6550ca70fdeaf33cb","evidence-official-790ab6283e5bb91474ed","evidence-official-dacb012b22c0a7de4880","evidence-official-87c70bddba97a72a5e5d","evidence-official-ba12a2d759aba6800ae6"],"source_locator":"GlycanGT paper Sections 2.1–2.4,2.9 and 4; official model-card licence metadata; primary paper Sections 2.1 and 3.1 versus README Model architecture / Training details; model/config_tokengt.py: size presets; publisher supplementary Table S2, Total Parameters row"},{"label":"Context limits","value":"The inspected model presets set graph-identifier dimensions and attention sizes, not a validated maximum glycan node count. Token length depends on both monosaccharides and linkages.","status":"unreported","source_ids":["evidence-official-10a6550ca70fdeaf33cb","evidence-official-790ab6283e5bb91474ed","evidence-official-dacb012b22c0a7de4880","evidence-official-87c70bddba97a72a5e5d","evidence-official-ba12a2d759aba6800ae6"],"source_locator":"GlycanGT paper Sections 2.1–2.4,2.9 and 4; official model-card licence metadata; primary paper Sections 2.1 and 3.1 versus README Model architecture / Training details; model/config_tokengt.py: size presets; publisher supplementary Table S2, Total Parameters row"},{"label":"Weights licence","value":"Apache-2.0 declared in the author-linked Akikitani295/GlycanGT model-card metadata.","status":"source_checked","source_ids":["evidence-official-10a6550ca70fdeaf33cb","evidence-official-790ab6283e5bb91474ed","evidence-official-dacb012b22c0a7de4880","evidence-official-87c70bddba97a72a5e5d","evidence-official-ba12a2d759aba6800ae6"],"source_locator":"GlycanGT paper Sections 2.1–2.4,2.9 and 4; official model-card licence metadata; primary paper Sections 2.1 and 3.1 versus README Model architecture / Training details; model/config_tokengt.py: size presets; publisher supplementary Table S2, Total Parameters row"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/matsui-lab/GlycanGT","status":"source_checked","source_ids":["evidence-official-10a6550ca70fdeaf33cb","evidence-official-790ab6283e5bb91474ed","evidence-official-dacb012b22c0a7de4880","evidence-official-87c70bddba97a72a5e5d","evidence-official-ba12a2d759aba6800ae6"],"source_locator":"GlycanGT paper Sections 2.1–2.4,2.9 and 4; official model-card licence metadata; primary paper Sections 2.1 and 3.1 versus README Model architecture / Training details; model/config_tokengt.py: size presets; publisher supplementary Table S2, Total Parameters row"},{"label":"Code licence","value":"Apache-2.0","status":"source_checked","source_ids":["evidence-official-9576e3936b5a16a999cf"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"Represents linkages explicitly and supports prediction of ambiguous glycan components.","source_ids":["evidence-official-10a6550ca70fdeaf33cb","evidence-official-790ab6283e5bb91474ed","evidence-official-dacb012b22c0a7de4880","evidence-official-87c70bddba97a72a5e5d","evidence-official-ba12a2d759aba6800ae6"],"source_locator":"GlycanGT paper Sections 2.1–2.4,2.9 and 4; official model-card licence metadata; primary paper Sections 2.1 and 3.1 versus README Model architecture / Training details; model/config_tokengt.py: size presets; publisher supplementary Table S2, Total Parameters row"}],"limitations":[{"text":"The training corpus excluded ambiguous entries. Downstream taxonomy, glycosylation and immunogenicity tasks require their own evaluation data and heads.","source_ids":["evidence-official-10a6550ca70fdeaf33cb","evidence-official-790ab6283e5bb91474ed","evidence-official-dacb012b22c0a7de4880","evidence-official-87c70bddba97a72a5e5d","evidence-official-ba12a2d759aba6800ae6"],"source_locator":"GlycanGT paper Sections 2.1–2.4,2.9 and 4; official model-card licence metadata; primary paper Sections 2.1 and 3.1 versus README Model architecture / Training details; model/config_tokengt.py: size presets; publisher supplementary Table S2, Total Parameters row"}],"diagram":{"title":"GlycanGT workflow","steps":["Glycan graph","Node and edge tokens","Graph transformer","Embedding or masked component"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-10a6550ca70fdeaf33cb","evidence-official-790ab6283e5bb91474ed","evidence-official-dacb012b22c0a7de4880","evidence-official-87c70bddba97a72a5e5d","evidence-official-ba12a2d759aba6800ae6"],"source_locator":"GlycanGT paper Sections 2.1–2.4,2.9 and 4; official model-card licence metadata; primary paper Sections 2.1 and 3.1 versus README Model architecture / Training details; model/config_tokengt.py: size presets; publisher supplementary Table S2, Total Parameters row"},"coverage":"limited","gaps":["Context limits: The inspected model presets set graph-identifier dimensions and attention sizes, not a validated maximum glycan node count. Token length depends on both monosaccharides and linkages."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}},"description":"Glycan graph transformer","facets":{"areas":["glycomics"]},"id":"discovery-model-glycangt","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-glycanml"}],"name":"GlycanGT","source_ids":["src-discovery-matsui-lab-glycangt"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-beeline"],"entity_level":"method","reported_name":"GRNBoost","version":null,"historical_missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"GRNBoost infers gene-regulatory networks with distributed gradient-boosted regression.","summary_source_ids":["evidence-official-099d109417c7fae96e96"],"summary_source_locator":"README.md: Introduction and License","sections":[{"title":"How it works","body":"GRNBoost infers gene-regulatory networks with distributed gradient-boosted regression. Spark pipeline implementing target-wise regression with XGBoost, following the GENIE3 inference blueprint. The documented inputs are gene-expression data and candidate regulatory genes. The output consists of predictive regulator-to-target links derived from the fitted regressions.","source_ids":["evidence-official-099d109417c7fae96e96"],"source_locator":"README.md: Introduction and License"},{"title":"Versions and reproducibility","body":"Spark/XGBoost GRNBoost implementation; do not substitute GRNBoost2 silently. Input gene matrix and distributed memory; no sequence token window.","source_ids":["evidence-official-099d109417c7fae96e96"],"source_locator":"README.md: Introduction and License"}],"facts":[{"label":"Model type","value":"Gradient-boosted regulatory-network inference","status":"source_checked","source_ids":["evidence-official-099d109417c7fae96e96"],"source_locator":"README.md: Introduction and License"},{"label":"Architecture","value":"Spark pipeline implementing target-wise regression with XGBoost, following the GENIE3 inference blueprint.","status":"source_checked","source_ids":["evidence-official-099d109417c7fae96e96"],"source_locator":"README.md: Introduction and License"},{"label":"Inputs","value":"Gene-expression data and candidate regulatory genes.","status":"source_checked","source_ids":["evidence-official-099d109417c7fae96e96"],"source_locator":"README.md: Introduction and License"},{"label":"Outputs","value":"Predictive regulator-to-target links derived from the fitted regressions.","status":"source_checked","source_ids":["evidence-official-099d109417c7fae96e96"],"source_locator":"README.md: Introduction and License"},{"label":"Parameters","value":"Dataset- and boosting-configuration-dependent.","status":"source_checked","source_ids":["evidence-official-099d109417c7fae96e96"],"source_locator":"README.md: Introduction and License"},{"label":"Known versions","value":"Spark/XGBoost GRNBoost implementation; do not substitute GRNBoost2 silently.","status":"source_checked","source_ids":["evidence-official-099d109417c7fae96e96"],"source_locator":"README.md: Introduction and License"},{"label":"Training data","value":"Fitted to the supplied expression data.","status":"source_checked","source_ids":["evidence-official-099d109417c7fae96e96"],"source_locator":"README.md: Introduction and License"},{"label":"Training cutoff","value":"Inapplicable to neural pretraining; reference-database and input-data dates must be recorded for each run.","status":"inapplicable","source_ids":["evidence-official-099d109417c7fae96e96"],"source_locator":"README.md: Introduction and License"},{"label":"Context limits","value":"Input gene matrix and distributed memory; no sequence token window.","status":"source_checked","source_ids":["evidence-official-099d109417c7fae96e96"],"source_locator":"README.md: Introduction and License"},{"label":"Weights licence","value":"No universal neural checkpoint; fitted regressors depend on the input dataset.","status":"source_checked","source_ids":["evidence-official-099d109417c7fae96e96"],"source_locator":"README.md: Introduction and License"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/aertslab/GRNBoost","status":"source_checked","source_ids":["evidence-official-099d109417c7fae96e96"],"source_locator":"README.md: Introduction and License"},{"label":"Code licence","value":"BSD-3-Clause","status":"source_checked","source_ids":["evidence-official-af4cadc1f0b4c69528d4"],"source_locator":"LICENSE.txt: licence text"}],"strengths":[{"text":"Distributes target-wise regression using Spark and replaces random forests with gradient boosting.","source_ids":["evidence-official-099d109417c7fae96e96"],"source_locator":"README.md: Introduction and License"}],"limitations":[{"text":"This repository describes GRNBoost, not every later GRNBoost2 implementation. Predictive links remain hypotheses about regulation.","source_ids":["evidence-official-099d109417c7fae96e96"],"source_locator":"README.md: Introduction and License"}],"diagram":{"title":"GRNBoost workflow","steps":["Expression and regulator set","Distributed target-wise boosting","Feature importance","Candidate regulatory network"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-099d109417c7fae96e96"],"source_locator":"README.md: Introduction and License"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}},"description":"Boosted-tree regulatory network inference","facets":{"areas":["biological-networks"]},"id":"discovery-model-grnboost","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-beeline"}],"name":"GRNBoost","source_ids":["src-discovery-aertslab-grnboost"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-cafa"],"entity_level":"method","reported_name":"HH-suite","version":null,"historical_missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"HH-suite searches for remote protein relationships using profile hidden Markov models.","summary_source_ids":["evidence-official-863f6c2e6e18152c2f8b"],"summary_source_locator":"README.md: opening, Available Databases and Usage","sections":[{"title":"How it works","body":"HH-suite searches for remote protein relationships using profile hidden Markov models. Pairwise profile-HMM alignment with iterative homolog search utilities such as HHblits. The documented inputs are protein sequence or alignment/profile and a selected reference profile database. The output consists of homolog hits, alignments and related search scores.","source_ids":["evidence-official-863f6c2e6e18152c2f8b"],"source_locator":"README.md: opening, Available Databases and Usage"},{"title":"Versions and reproducibility","body":"HH-suite3; README includes v3.3.0 binaries. Query/alignment size and database settings, not a learned token context.","source_ids":["evidence-official-863f6c2e6e18152c2f8b"],"source_locator":"README.md: opening, Available Databases and Usage"}],"facts":[{"label":"Model type","value":"Profile-HMM sequence search software","status":"source_checked","source_ids":["evidence-official-863f6c2e6e18152c2f8b"],"source_locator":"README.md: opening, Available Databases and Usage"},{"label":"Architecture","value":"Pairwise profile-HMM alignment with iterative homolog search utilities such as HHblits.","status":"source_checked","source_ids":["evidence-official-863f6c2e6e18152c2f8b"],"source_locator":"README.md: opening, Available Databases and Usage"},{"label":"Inputs","value":"Protein sequence or alignment/profile and a selected reference profile database.","status":"source_checked","source_ids":["evidence-official-863f6c2e6e18152c2f8b"],"source_locator":"README.md: opening, Available Databases and Usage"},{"label":"Outputs","value":"Homolog hits, alignments and related search scores.","status":"source_checked","source_ids":["evidence-official-863f6c2e6e18152c2f8b"],"source_locator":"README.md: opening, Available Databases and Usage"},{"label":"Parameters","value":"Profile probabilities and search settings; not a fixed neural parameter count.","status":"source_checked","source_ids":["evidence-official-863f6c2e6e18152c2f8b"],"source_locator":"README.md: opening, Available Databases and Usage"},{"label":"Known versions","value":"HH-suite3; README includes v3.3.0 binaries.","status":"source_checked","source_ids":["evidence-official-863f6c2e6e18152c2f8b"],"source_locator":"README.md: opening, Available Databases and Usage"},{"label":"Training data","value":"Profiles/reference databases rather than neural pretraining; examples include Uniclust30, BFD and PDB70.","status":"source_checked","source_ids":["evidence-official-863f6c2e6e18152c2f8b"],"source_locator":"README.md: opening, Available Databases and Usage"},{"label":"Training cutoff","value":"Inapplicable to neural pretraining; reference-database and input-data dates must be recorded for each run.","status":"inapplicable","source_ids":["evidence-official-863f6c2e6e18152c2f8b"],"source_locator":"README.md: opening, Available Databases and Usage"},{"label":"Context limits","value":"Query/alignment size and database settings, not a learned token context.","status":"source_checked","source_ids":["evidence-official-863f6c2e6e18152c2f8b"],"source_locator":"README.md: opening, Available Databases and Usage"},{"label":"Weights licence","value":"Inapplicable: reference profiles/databases are the relevant artifacts.","status":"inapplicable","source_ids":["evidence-official-863f6c2e6e18152c2f8b"],"source_locator":"README.md: opening, Available Databases and Usage"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/soedinglab/hh-suite","status":"source_checked","source_ids":["evidence-official-863f6c2e6e18152c2f8b"],"source_locator":"README.md: opening, Available Databases and Usage"},{"label":"Code licence","value":"GPL-3.0; inspect the pinned licence and any file-specific terms.","status":"source_checked","source_ids":["evidence-official-00ee97a5586392ae2348"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"Can search at the profile level and construct alignments of homologous sequences.","source_ids":["evidence-official-863f6c2e6e18152c2f8b"],"source_locator":"README.md: opening, Available Databases and Usage"}],"limitations":[{"text":"Search results depend on database coverage and iteration/settings. Profile similarity is evidence of sequence relationships rather than a direct functional assay.","source_ids":["evidence-official-863f6c2e6e18152c2f8b"],"source_locator":"README.md: opening, Available Databases and Usage"}],"diagram":{"title":"HH-suite workflow","steps":["Query sequence or alignment","Profile HMM","Reference profile search","Homologs and alignment"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-863f6c2e6e18152c2f8b"],"source_locator":"README.md: opening, Available Databases and Usage"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}},"description":"Profile hidden Markov sequence search","facets":{"areas":["protein-function"]},"id":"discovery-model-hh-suite","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-cafa"}],"name":"HH-suite","source_ids":["src-discovery-soedinglab-hh-suite"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":[],"entity_level":"method","reported_name":"HUMAnN","version":null,"historical_missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"HUMAnN profiles microbial gene functions and pathways from metagenomic or metatranscriptomic reads.","summary_source_ids":["evidence-official-10f58e320879def8b102"],"summary_source_locator":"readme.md: description, Main workflow, Download the databases and normalization documentation","sections":[{"title":"How it works","body":"HUMAnN profiles microbial gene functions and pathways from metagenomic or metatranscriptomic reads. Reference-based functional profiling pipeline with nucleotide and translated search against configured databases. The documented inputs are short DNA/RNA reads or supported preprocessed alignment/profile inputs. The output consists of gene-family and pathway abundance/coverage tables, including organism-stratified outputs.","source_ids":["evidence-official-10f58e320879def8b102"],"source_locator":"readme.md: description, Main workflow, Download the databases and normalization documentation"},{"title":"Versions and reproducibility","body":"HUMAnN 3.0 paper and implementation; nucleotide/protein database versions must be preserved separately. Read files and search-database size; no fixed model context.","source_ids":["evidence-official-10f58e320879def8b102"],"source_locator":"readme.md: description, Main workflow, Download the databases and normalization documentation"}],"facts":[{"label":"Model type","value":"Metagenomic functional profiling pipeline","status":"source_checked","source_ids":["evidence-official-10f58e320879def8b102"],"source_locator":"readme.md: description, Main workflow, Download the databases and normalization documentation"},{"label":"Architecture","value":"Reference-based functional profiling pipeline with nucleotide and translated search against configured databases.","status":"source_checked","source_ids":["evidence-official-10f58e320879def8b102"],"source_locator":"readme.md: description, Main workflow, Download the databases and normalization documentation"},{"label":"Inputs","value":"Short DNA/RNA reads or supported preprocessed alignment/profile inputs.","status":"source_checked","source_ids":["evidence-official-10f58e320879def8b102"],"source_locator":"readme.md: description, Main workflow, Download the databases and normalization documentation"},{"label":"Outputs","value":"Gene-family and pathway abundance/coverage tables, including organism-stratified outputs.","status":"source_checked","source_ids":["evidence-official-10f58e320879def8b102"],"source_locator":"readme.md: description, Main workflow, Download the databases and normalization documentation"},{"label":"Parameters","value":"Inapplicable as a neural parameter count.","status":"inapplicable","source_ids":["evidence-official-10f58e320879def8b102"],"source_locator":"readme.md: description, Main workflow, Download the databases and normalization documentation"},{"label":"Known versions","value":"HUMAnN 3.0 paper and implementation; nucleotide/protein database versions must be preserved separately.","status":"source_checked","source_ids":["evidence-official-10f58e320879def8b102"],"source_locator":"readme.md: description, Main workflow, Download the databases and normalization documentation"},{"label":"Training data","value":"ChocoPhlAn and translated protein-search databases are analysis resources, not a universal neural training corpus.","status":"source_checked","source_ids":["evidence-official-10f58e320879def8b102"],"source_locator":"readme.md: description, Main workflow, Download the databases and normalization documentation"},{"label":"Training cutoff","value":"Inapplicable to neural pretraining; reference-database and input-data dates must be recorded for each run.","status":"inapplicable","source_ids":["evidence-official-10f58e320879def8b102"],"source_locator":"readme.md: description, Main workflow, Download the databases and normalization documentation"},{"label":"Context limits","value":"Read files and search-database size; no fixed model context.","status":"source_checked","source_ids":["evidence-official-10f58e320879def8b102"],"source_locator":"readme.md: description, Main workflow, Download the databases and normalization documentation"},{"label":"Weights licence","value":"Inapplicable: no neural checkpoint for the core profiling pipeline.","status":"inapplicable","source_ids":["evidence-official-10f58e320879def8b102"],"source_locator":"readme.md: description, Main workflow, Download the databases and normalization documentation"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/biobakery/humann","status":"source_checked","source_ids":["evidence-official-10f58e320879def8b102"],"source_locator":"readme.md: description, Main workflow, Download the databases and normalization documentation"},{"label":"Code licence","value":"MIT","status":"source_checked","source_ids":["evidence-official-f29710879cf7a4438b12"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"Connects community sequencing data to interpretable functional and pathway profiles.","source_ids":["evidence-official-10f58e320879def8b102"],"source_locator":"readme.md: description, Main workflow, Download the databases and normalization documentation"}],"limitations":[{"text":"Reference coverage and normalization affect interpretation. The manual clarifies that CPM means copies per million rather than unnormalized counts per million.","source_ids":["evidence-official-10f58e320879def8b102"],"source_locator":"readme.md: description, Main workflow, Download the databases and normalization documentation"}],"diagram":{"title":"HUMAnN workflow","steps":["Community reads","Reference searches","Gene-family quantification","Pathway profiles"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-10f58e320879def8b102"],"source_locator":"readme.md: description, Main workflow, Download the databases and normalization documentation"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}},"description":"Microbial functional profiling","facets":{"areas":["microbiome"]},"id":"discovery-model-humann","kind":"model","links":[],"name":"HUMAnN","source_ids":["src-discovery-biobakery-humann"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-cami"],"entity_level":"method","reported_name":"Kraken 2","version":null,"historical_missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"Kraken 2 assigns taxonomic labels to sequence reads by consulting a reference-derived minimizer database.","summary_source_ids":["evidence-official-40f46dc4b9fcc306773b","evidence-official-33fe0374c8ebbc2c7294","evidence-official-9ba2b6dc494c7b6df951","evidence-official-1f756fefa823dd402b0f"],"summary_source_locator":"docs/MANUAL.markdown: Introduction, Standard Kraken 2 Database, Classification and Confidence scoring","sections":[{"title":"How it works","body":"Kraken 2 breaks query sequences into k-mers and looks up selected minimizers in a compact hash table. Each stored minimizer is associated with a lowest-common-ancestor taxonomic label. The classifier combines that evidence to assign a taxon; the reference database and confidence settings are therefore part of the evaluated procedure.","source_ids":["evidence-official-40f46dc4b9fcc306773b","evidence-official-33fe0374c8ebbc2c7294","evidence-official-9ba2b6dc494c7b6df951","evidence-official-1f756fefa823dd402b0f"],"source_locator":"docs/MANUAL.markdown: Introduction, Standard Kraken 2 Database, Classification and Confidence scoring"},{"title":"Versions and reproducibility","body":"Kraken 2 is a rewrite of Kraken 1 and is not backwards compatible. Read/contig input, not a learned fixed token window.","source_ids":["evidence-official-40f46dc4b9fcc306773b","evidence-official-33fe0374c8ebbc2c7294","evidence-official-9ba2b6dc494c7b6df951","evidence-official-1f756fefa823dd402b0f"],"source_locator":"docs/MANUAL.markdown: Introduction, Standard Kraken 2 Database, Classification and Confidence scoring"}],"facts":[{"label":"Model type","value":"Minimizer-based taxonomic classifier","status":"source_checked","source_ids":["evidence-official-40f46dc4b9fcc306773b","evidence-official-33fe0374c8ebbc2c7294","evidence-official-9ba2b6dc494c7b6df951","evidence-official-1f756fefa823dd402b0f"],"source_locator":"docs/MANUAL.markdown: Introduction, Standard Kraken 2 Database, Classification and Confidence scoring"},{"label":"Architecture","value":"Minimizer-based sequence classification using a compact hash table and lowest-common-ancestor taxonomy assignments.","status":"source_checked","source_ids":["evidence-official-40f46dc4b9fcc306773b","evidence-official-33fe0374c8ebbc2c7294","evidence-official-9ba2b6dc494c7b6df951","evidence-official-1f756fefa823dd402b0f"],"source_locator":"docs/MANUAL.markdown: Introduction, Standard Kraken 2 Database, Classification and Confidence scoring"},{"label":"Inputs","value":"DNA reads or, in translated-search mode, sequences searched against an amino-acid database.","status":"source_checked","source_ids":["evidence-official-40f46dc4b9fcc306773b","evidence-official-33fe0374c8ebbc2c7294","evidence-official-9ba2b6dc494c7b6df951","evidence-official-1f756fefa823dd402b0f"],"source_locator":"docs/MANUAL.markdown: Introduction, Standard Kraken 2 Database, Classification and Confidence scoring"},{"label":"Outputs","value":"Per-read taxonomic assignments and aggregate classification reports.","status":"source_checked","source_ids":["evidence-official-40f46dc4b9fcc306773b","evidence-official-33fe0374c8ebbc2c7294","evidence-official-9ba2b6dc494c7b6df951","evidence-official-1f756fefa823dd402b0f"],"source_locator":"docs/MANUAL.markdown: Introduction, Standard Kraken 2 Database, Classification and Confidence scoring"},{"label":"Parameters","value":"Inapplicable as a neural parameter total; k-mer/minimizer length, confidence and database choices are algorithm settings.","status":"inapplicable","source_ids":["evidence-official-40f46dc4b9fcc306773b","evidence-official-33fe0374c8ebbc2c7294","evidence-official-9ba2b6dc494c7b6df951","evidence-official-1f756fefa823dd402b0f"],"source_locator":"docs/MANUAL.markdown: Introduction, Standard Kraken 2 Database, Classification and Confidence scoring"},{"label":"Known versions","value":"Kraken 2 is a rewrite of Kraken 1 and is not backwards compatible.","status":"source_checked","source_ids":["evidence-official-40f46dc4b9fcc306773b","evidence-official-33fe0374c8ebbc2c7294","evidence-official-9ba2b6dc494c7b6df951","evidence-official-1f756fefa823dd402b0f"],"source_locator":"docs/MANUAL.markdown: Introduction, Standard Kraken 2 Database, Classification and Confidence scoring"},{"label":"Training data","value":"Not neural pretraining: build a database from selected reference sequences and taxonomy.","status":"source_checked","source_ids":["evidence-official-40f46dc4b9fcc306773b","evidence-official-33fe0374c8ebbc2c7294","evidence-official-9ba2b6dc494c7b6df951","evidence-official-1f756fefa823dd402b0f"],"source_locator":"docs/MANUAL.markdown: Introduction, Standard Kraken 2 Database, Classification and Confidence scoring"},{"label":"Training cutoff","value":"Inapplicable to neural pretraining; reference-database and input-data dates must be recorded for each run.","status":"inapplicable","source_ids":["evidence-official-40f46dc4b9fcc306773b","evidence-official-33fe0374c8ebbc2c7294","evidence-official-9ba2b6dc494c7b6df951","evidence-official-1f756fefa823dd402b0f"],"source_locator":"docs/MANUAL.markdown: Introduction, Standard Kraken 2 Database, Classification and Confidence scoring"},{"label":"Context limits","value":"Read/contig input, not a learned fixed token window.","status":"source_checked","source_ids":["evidence-official-40f46dc4b9fcc306773b","evidence-official-33fe0374c8ebbc2c7294","evidence-official-9ba2b6dc494c7b6df951","evidence-official-1f756fefa823dd402b0f"],"source_locator":"docs/MANUAL.markdown: Introduction, Standard Kraken 2 Database, Classification and Confidence scoring"},{"label":"Weights licence","value":"Inapplicable to this classifier: database contents and their licences replace neural weights.","status":"inapplicable","source_ids":["evidence-official-40f46dc4b9fcc306773b","evidence-official-33fe0374c8ebbc2c7294","evidence-official-9ba2b6dc494c7b6df951","evidence-official-1f756fefa823dd402b0f"],"source_locator":"docs/MANUAL.markdown: Introduction, Standard Kraken 2 Database, Classification and Confidence scoring"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/DerrickWood/kraken2","status":"source_checked","source_ids":["evidence-official-40f46dc4b9fcc306773b","evidence-official-33fe0374c8ebbc2c7294","evidence-official-9ba2b6dc494c7b6df951","evidence-official-1f756fefa823dd402b0f"],"source_locator":"docs/MANUAL.markdown: Introduction, Standard Kraken 2 Database, Classification and Confidence scoring"},{"label":"Code licence","value":"MIT","status":"source_checked","source_ids":["evidence-official-bf1a5e03cd84873f4b04"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"A reference-based procedural comparator with explicit database construction and confidence settings.","source_ids":["evidence-official-40f46dc4b9fcc306773b","evidence-official-33fe0374c8ebbc2c7294","evidence-official-9ba2b6dc494c7b6df951","evidence-official-1f756fefa823dd402b0f"],"source_locator":"docs/MANUAL.markdown: Introduction, Standard Kraken 2 Database, Classification and Confidence scoring"}],"limitations":[{"text":"Classification depends on the reference database, taxonomy version and minimizer configuration. Compact hashing can introduce false matches; software version alone does not identify a reproducible classifier.","source_ids":["evidence-official-40f46dc4b9fcc306773b","evidence-official-33fe0374c8ebbc2c7294","evidence-official-9ba2b6dc494c7b6df951","evidence-official-1f756fefa823dd402b0f"],"source_locator":"docs/MANUAL.markdown: Introduction, Standard Kraken 2 Database, Classification and Confidence scoring"}],"diagram":{"title":"Kraken 2 workflow","steps":["Reference genomes and taxonomy","Minimizer database","Read minimizer lookup","Taxonomic assignment"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-40f46dc4b9fcc306773b","evidence-official-33fe0374c8ebbc2c7294","evidence-official-9ba2b6dc494c7b6df951","evidence-official-1f756fefa823dd402b0f"],"source_locator":"docs/MANUAL.markdown: Introduction, Standard Kraken 2 Database, Classification and Confidence scoring"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}},"description":"Sequence-based metagenomic classification","facets":{"areas":["microbiome"]},"id":"discovery-model-kraken-2","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-cami"}],"name":"Kraken 2","source_ids":["src-discovery-derrickwood-kraken2"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":[],"entity_level":"method","reported_name":"LipidBlast","version":null,"historical_missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"LipidBlast is an in-silico tandem mass-spectral library used to annotate lipids.","summary_source_ids":["evidence-official-fadad0bf451a225696d1"],"summary_source_locator":"Official LipidBlast page: Short Introduction, Download, Picture service and FAQ","sections":[{"title":"How it works","body":"LipidBlast is an in-silico tandem mass-spectral library used to annotate lipids. Rule/template-generated lipid fragmentation library queried by a compatible mass-spectral search engine. The documented inputs are LC-MS/MS spectra or, in the documented accurate-mass-only workflow, precursor masses. The output consists of candidate lipid annotations from spectral or mass matching.","source_ids":["evidence-official-fadad0bf451a225696d1"],"source_locator":"Official LipidBlast page: Short Introduction, Download, Picture service and FAQ"},{"title":"Versions and reproducibility","body":"Original LipidBlast library and later MS-DIAL integration are distinct releases. Mass-spectral peaks and library/search settings; no neural token window.","source_ids":["evidence-official-fadad0bf451a225696d1"],"source_locator":"Official LipidBlast page: Short Introduction, Download, Picture service and FAQ"}],"facts":[{"label":"Model type","value":"Rule-based in-silico lipid spectral library","status":"source_checked","source_ids":["evidence-official-fadad0bf451a225696d1"],"source_locator":"Official LipidBlast page: Short Introduction, Download, Picture service and FAQ"},{"label":"Architecture","value":"Rule/template-generated lipid fragmentation library queried by a compatible mass-spectral search engine.","status":"source_checked","source_ids":["evidence-official-fadad0bf451a225696d1"],"source_locator":"Official LipidBlast page: Short Introduction, Download, Picture service and FAQ"},{"label":"Inputs","value":"LC-MS/MS spectra or, in the documented accurate-mass-only workflow, precursor masses.","status":"source_checked","source_ids":["evidence-official-fadad0bf451a225696d1"],"source_locator":"Official LipidBlast page: Short Introduction, Download, Picture service and FAQ"},{"label":"Outputs","value":"Candidate lipid annotations from spectral or mass matching.","status":"source_checked","source_ids":["evidence-official-fadad0bf451a225696d1"],"source_locator":"Official LipidBlast page: Short Introduction, Download, Picture service and FAQ"},{"label":"Parameters","value":"Inapplicable as a neural parameter count.","status":"inapplicable","source_ids":["evidence-official-fadad0bf451a225696d1"],"source_locator":"Official LipidBlast page: Short Introduction, Download, Picture service and FAQ"},{"label":"Known versions","value":"Original LipidBlast library and later MS-DIAL integration are distinct releases.","status":"source_checked","source_ids":["evidence-official-fadad0bf451a225696d1"],"source_locator":"Official LipidBlast page: Short Introduction, Download, Picture service and FAQ"},{"label":"Training data","value":"Expert fragmentation templates and reference evidence, not neural pretraining.","status":"source_checked","source_ids":["evidence-official-fadad0bf451a225696d1"],"source_locator":"Official LipidBlast page: Short Introduction, Download, Picture service and FAQ"},{"label":"Training cutoff","value":"Inapplicable to neural pretraining; reference-database and input-data dates must be recorded for each run.","status":"inapplicable","source_ids":["evidence-official-fadad0bf451a225696d1"],"source_locator":"Official LipidBlast page: Short Introduction, Download, Picture service and FAQ"},{"label":"Context limits","value":"Mass-spectral peaks and library/search settings; no neural token window.","status":"source_checked","source_ids":["evidence-official-fadad0bf451a225696d1"],"source_locator":"Official LipidBlast page: Short Introduction, Download, Picture service and FAQ"},{"label":"Weights licence","value":"Inapplicable: library and software terms must be checked separately. The page only explicitly grants CC-BY for parts of the publication software supplement.","status":"inapplicable","source_ids":["evidence-official-fadad0bf451a225696d1"],"source_locator":"Official LipidBlast page: Short Introduction, Download, Picture service and FAQ"},{"label":"Access","value":"Official project documentation and implementation: https://fiehnlab.ucdavis.edu/projects/LipidBlast/","status":"source_checked","source_ids":["evidence-official-fadad0bf451a225696d1"],"source_locator":"Official LipidBlast page: Short Introduction, Download, Picture service and FAQ"},{"label":"Code licence","value":"The inspected LipidBlast project page does not establish separate software/library redistribution terms. The accompanying article licence is not substituted for those terms.","status":"unreported","source_ids":["evidence-official-fadad0bf451a225696d1"],"source_locator":"page.html: inspected official source"}],"strengths":[{"text":"Supports library search across multiple instrument types and provides inspectable reference spectra/templates.","source_ids":["evidence-official-fadad0bf451a225696d1"],"source_locator":"Official LipidBlast page: Short Introduction, Download, Picture service and FAQ"}],"limitations":[{"text":"The project notes missing lipid classes and ambiguous accurate-mass matches. Identification by mass alone cannot resolve isobars; it is not intended for GC-MS.","source_ids":["evidence-official-fadad0bf451a225696d1"],"source_locator":"Official LipidBlast page: Short Introduction, Download, Picture service and FAQ"}],"diagram":{"title":"LipidBlast workflow","steps":["Lipid fragmentation templates","In-silico library","Experimental-spectrum search","Candidate lipid annotation"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-fadad0bf451a225696d1"],"source_locator":"Official LipidBlast page: Short Introduction, Download, Picture service and FAQ"},"coverage":"limited","gaps":["Code licence: The inspected LipidBlast project page does not establish separate software/library redistribution terms. The accompanying article licence is not substituted for those terms."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}},"description":"Rule-based lipid fragmentation library matching","facets":{"areas":["lipidomics"]},"id":"discovery-model-lipidblast","kind":"model","links":[],"name":"LipidBlast","source_ids":["src-discovery-lipidblast"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":[],"entity_level":"method","reported_name":"LipidFinder","version":null,"historical_missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"LipidFinder filters aligned LC-MS features and assigns putative lipid identities using reference databases.","summary_source_ids":["evidence-official-95a620349435ae6651c5","evidence-official-3759a2f34cb49263eb43"],"summary_source_locator":"Official LipidFinder page: workflow description, applicability warning and source-code link","sections":[{"title":"How it works","body":"LipidFinder filters aligned LC-MS features and assigns putative lipid identities using reference databases. Configurable filtering and database-search workflow over high-resolution LC-MS features. The documented inputs are pre-aligned high-resolution LC-MS measurements, including chromatographic information. The output consists of filtered lipid-like features and putative lipid classifications.","source_ids":["evidence-official-95a620349435ae6651c5","evidence-official-3759a2f34cb49263eb43"],"source_locator":"Official LipidFinder page: workflow description, applicability warning and source-code link"},{"title":"Versions and reproducibility","body":"LipidFinder original workflow and LipidFinder 2.0 are distinguished in the official page citations. Chromatographic features and mass-resolution constraints, not sequence tokens.","source_ids":["evidence-official-95a620349435ae6651c5","evidence-official-3759a2f34cb49263eb43"],"source_locator":"Official LipidFinder page: workflow description, applicability warning and source-code link"}],"facts":[{"label":"Model type","value":"Lipidomics preprocessing and identification workflow","status":"source_checked","source_ids":["evidence-official-95a620349435ae6651c5","evidence-official-3759a2f34cb49263eb43"],"source_locator":"Official LipidFinder page: workflow description, applicability warning and source-code link"},{"label":"Architecture","value":"Configurable filtering and database-search workflow over high-resolution LC-MS features.","status":"source_checked","source_ids":["evidence-official-95a620349435ae6651c5","evidence-official-3759a2f34cb49263eb43"],"source_locator":"Official LipidFinder page: workflow description, applicability warning and source-code link"},{"label":"Inputs","value":"Pre-aligned high-resolution LC-MS measurements, including chromatographic information.","status":"source_checked","source_ids":["evidence-official-95a620349435ae6651c5","evidence-official-3759a2f34cb49263eb43"],"source_locator":"Official LipidFinder page: workflow description, applicability warning and source-code link"},{"label":"Outputs","value":"Filtered lipid-like features and putative lipid classifications.","status":"source_checked","source_ids":["evidence-official-95a620349435ae6651c5","evidence-official-3759a2f34cb49263eb43"],"source_locator":"Official LipidFinder page: workflow description, applicability warning and source-code link"},{"label":"Parameters","value":"Inapplicable as a neural parameter count; filtering tolerances and database choices are configuration.","status":"inapplicable","source_ids":["evidence-official-95a620349435ae6651c5","evidence-official-3759a2f34cb49263eb43"],"source_locator":"Official LipidFinder page: workflow description, applicability warning and source-code link"},{"label":"Known versions","value":"LipidFinder original workflow and LipidFinder 2.0 are distinguished in the official page citations.","status":"source_checked","source_ids":["evidence-official-95a620349435ae6651c5","evidence-official-3759a2f34cb49263eb43"],"source_locator":"Official LipidFinder page: workflow description, applicability warning and source-code link"},{"label":"Training data","value":"No neural pretraining; reference databases and user settings define the workflow.","status":"source_checked","source_ids":["evidence-official-95a620349435ae6651c5","evidence-official-3759a2f34cb49263eb43"],"source_locator":"Official LipidFinder page: workflow description, applicability warning and source-code link"},{"label":"Training cutoff","value":"Inapplicable to neural pretraining; reference-database and input-data dates must be recorded for each run.","status":"inapplicable","source_ids":["evidence-official-95a620349435ae6651c5","evidence-official-3759a2f34cb49263eb43"],"source_locator":"Official LipidFinder page: workflow description, applicability warning and source-code link"},{"label":"Context limits","value":"Chromatographic features and mass-resolution constraints, not sequence tokens.","status":"source_checked","source_ids":["evidence-official-95a620349435ae6651c5","evidence-official-3759a2f34cb49263eb43"],"source_locator":"Official LipidFinder page: workflow description, applicability warning and source-code link"},{"label":"Weights licence","value":"Inapplicable: no neural model checkpoint.","status":"inapplicable","source_ids":["evidence-official-95a620349435ae6651c5","evidence-official-3759a2f34cb49263eb43"],"source_locator":"Official LipidFinder page: workflow description, applicability warning and source-code link"},{"label":"Access","value":"Official project documentation and implementation: https://www.lipidmaps.org/resources/tools/lipidfinder/","status":"source_checked","source_ids":["evidence-official-95a620349435ae6651c5","evidence-official-3759a2f34cb49263eb43"],"source_locator":"Official LipidFinder page: workflow description, applicability warning and source-code link"},{"label":"Code licence","value":"MIT","status":"source_checked","source_ids":["evidence-official-5631fb75aa8e5a7a89e1"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"Separates likely lipids from contaminants, adducts and noise before database annotation.","source_ids":["evidence-official-95a620349435ae6651c5","evidence-official-3759a2f34cb49263eb43"],"source_locator":"Official LipidFinder page: workflow description, applicability warning and source-code link"}],"limitations":[{"text":"The official page explicitly excludes shotgun lipidomics, MS/MS and low-resolution data from the described workflow. Putative annotations need separate identification evidence.","source_ids":["evidence-official-95a620349435ae6651c5","evidence-official-3759a2f34cb49263eb43"],"source_locator":"Official LipidFinder page: workflow description, applicability warning and source-code link"}],"diagram":{"title":"LipidFinder workflow","steps":["Aligned LC-MS features","Filtering and contaminant handling","Database search","Putative lipid classes"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-95a620349435ae6651c5","evidence-official-3759a2f34cb49263eb43"],"source_locator":"Official LipidFinder page: workflow description, applicability warning and source-code link"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}},"description":"LC-MS lipid feature filtering and annotation","facets":{"areas":["lipidomics"]},"id":"discovery-model-lipidfinder","kind":"model","links":[],"name":"LipidFinder","source_ids":["src-discovery-lipidfinder"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-massspecgym"],"entity_level":"method","reported_name":"matchms","version":null,"historical_missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"matchms provides reproducible processing and comparison of tandem mass spectra.","summary_source_ids":["evidence-official-e508b40498c3a3ba53d5"],"summary_source_locator":"README.rst: introductory description and License","sections":[{"title":"How it works","body":"matchms provides reproducible processing and comparison of tandem mass spectra. Modular import, metadata/peak processing and pairwise-similarity pipeline; classical cosine and external learned measures are distinct options. The documented inputs are MS/MS spectra in supported formats such as mzML, mzXML, MSP, MGF and JSON. The output consists of processed spectra and pairwise similarity scores.","source_ids":["evidence-official-e508b40498c3a3ba53d5"],"source_locator":"README.rst: introductory description and License"},{"title":"Versions and reproducibility","body":"matchms version plus the exact filter/similarity pipeline. Spectral peaks and workflow memory requirements; no universal token context.","source_ids":["evidence-official-e508b40498c3a3ba53d5"],"source_locator":"README.rst: introductory description and License"}],"facts":[{"label":"Model type","value":"Mass-spectral processing and similarity software","status":"source_checked","source_ids":["evidence-official-e508b40498c3a3ba53d5"],"source_locator":"README.rst: introductory description and License"},{"label":"Architecture","value":"Modular import, metadata/peak processing and pairwise-similarity pipeline; classical cosine and external learned measures are distinct options.","status":"source_checked","source_ids":["evidence-official-e508b40498c3a3ba53d5"],"source_locator":"README.rst: introductory description and License"},{"label":"Inputs","value":"MS/MS spectra in supported formats such as mzML, mzXML, MSP, MGF and JSON.","status":"source_checked","source_ids":["evidence-official-e508b40498c3a3ba53d5"],"source_locator":"README.rst: introductory description and License"},{"label":"Outputs","value":"Processed spectra and pairwise similarity scores.","status":"source_checked","source_ids":["evidence-official-e508b40498c3a3ba53d5"],"source_locator":"README.rst: introductory description and License"},{"label":"Parameters","value":"Inapplicable for the framework; external learned similarity plugins have separate parameters.","status":"inapplicable","source_ids":["evidence-official-e508b40498c3a3ba53d5"],"source_locator":"README.rst: introductory description and License"},{"label":"Known versions","value":"matchms version plus the exact filter/similarity pipeline.","status":"source_checked","source_ids":["evidence-official-e508b40498c3a3ba53d5"],"source_locator":"README.rst: introductory description and License"},{"label":"Training data","value":"No universal pretraining; selected learned plugins must provide their own training provenance.","status":"source_checked","source_ids":["evidence-official-e508b40498c3a3ba53d5"],"source_locator":"README.rst: introductory description and License"},{"label":"Training cutoff","value":"Inapplicable to neural pretraining; reference-database and input-data dates must be recorded for each run.","status":"inapplicable","source_ids":["evidence-official-e508b40498c3a3ba53d5"],"source_locator":"README.rst: introductory description and License"},{"label":"Context limits","value":"Spectral peaks and workflow memory requirements; no universal token context.","status":"source_checked","source_ids":["evidence-official-e508b40498c3a3ba53d5"],"source_locator":"README.rst: introductory description and License"},{"label":"Weights licence","value":"Inapplicable to core classical processing; external learned plugins have separate weight terms.","status":"inapplicable","source_ids":["evidence-official-e508b40498c3a3ba53d5"],"source_locator":"README.rst: introductory description and License"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/matchms/matchms","status":"source_checked","source_ids":["evidence-official-e508b40498c3a3ba53d5"],"source_locator":"README.rst: introductory description and License"},{"label":"Code licence","value":"Apache-2.0","status":"source_checked","source_ids":["evidence-official-71e8ebc066164f0ce44e"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"Separates preprocessing from the similarity measure and supports custom metrics and sparse result storage.","source_ids":["evidence-official-e508b40498c3a3ba53d5"],"source_locator":"README.rst: introductory description and License"}],"limitations":[{"text":"matchms is a framework, not one universal scoring model. Filter order, thresholds and selected similarity function must be recorded.","source_ids":["evidence-official-e508b40498c3a3ba53d5"],"source_locator":"README.rst: introductory description and License"}],"diagram":{"title":"matchms workflow","steps":["Imported MS/MS spectra","Metadata and peak processing","Selected similarity function","Similarity matrix"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-e508b40498c3a3ba53d5"],"source_locator":"README.rst: introductory description and License"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}},"description":"Mass spectral processing and similarity matching","facets":{"areas":["metabolomics"]},"id":"discovery-model-matchms","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-massspecgym"}],"name":"matchms","source_ids":["src-discovery-matchms-matchms"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-cami"],"entity_level":"method","reported_name":"MetaPhlAn","version":null,"historical_missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"MetaPhlAn profiles microbial community composition from shotgun metagenomic reads using clade-specific marker genes.","summary_source_ids":["evidence-official-72fdaed158e4ff851fbe","evidence-official-37f825d1226e59cbbfbf"],"summary_source_locator":"README.md: description; MetaPhlAn 4 paper: Building the expanded SGB catalog and Discussion","sections":[{"title":"How it works","body":"MetaPhlAn profiles microbial community composition from shotgun metagenomic reads using clade-specific marker genes. Reference marker-gene profiling; MetaPhlAn 4 organizes reference and metagenome-assembled genomes into species-level genome bins. The documented inputs are shotgun metagenomic reads and a selected MetaPhlAn marker database. The output consists of taxonomic relative-abundance profiles; StrainPhlAn is a separate strain-level analysis.","source_ids":["evidence-official-72fdaed158e4ff851fbe","evidence-official-37f825d1226e59cbbfbf"],"source_locator":"README.md: description; MetaPhlAn 4 paper: Building the expanded SGB catalog and Discussion"},{"title":"Versions and reproducibility","body":"MetaPhlAn 4 paper and 4.2-linked current documentation; database version is a separate reproducibility requirement. Shotgun reads; no fixed neural token context.","source_ids":["evidence-official-72fdaed158e4ff851fbe","evidence-official-37f825d1226e59cbbfbf"],"source_locator":"README.md: description; MetaPhlAn 4 paper: Building the expanded SGB catalog and Discussion"}],"facts":[{"label":"Model type","value":"Marker-based taxonomic profiling","status":"source_checked","source_ids":["evidence-official-72fdaed158e4ff851fbe","evidence-official-37f825d1226e59cbbfbf"],"source_locator":"README.md: description; MetaPhlAn 4 paper: Building the expanded SGB catalog and Discussion"},{"label":"Architecture","value":"Reference marker-gene profiling; MetaPhlAn 4 organizes reference and metagenome-assembled genomes into species-level genome bins.","status":"source_checked","source_ids":["evidence-official-72fdaed158e4ff851fbe","evidence-official-37f825d1226e59cbbfbf"],"source_locator":"README.md: description; MetaPhlAn 4 paper: Building the expanded SGB catalog and Discussion"},{"label":"Inputs","value":"Shotgun metagenomic reads and a selected MetaPhlAn marker database.","status":"source_checked","source_ids":["evidence-official-72fdaed158e4ff851fbe","evidence-official-37f825d1226e59cbbfbf"],"source_locator":"README.md: description; MetaPhlAn 4 paper: Building the expanded SGB catalog and Discussion"},{"label":"Outputs","value":"Taxonomic relative-abundance profiles; StrainPhlAn is a separate strain-level analysis.","status":"source_checked","source_ids":["evidence-official-72fdaed158e4ff851fbe","evidence-official-37f825d1226e59cbbfbf"],"source_locator":"README.md: description; MetaPhlAn 4 paper: Building the expanded SGB catalog and Discussion"},{"label":"Parameters","value":"Inapplicable as a neural parameter count.","status":"inapplicable","source_ids":["evidence-official-72fdaed158e4ff851fbe","evidence-official-37f825d1226e59cbbfbf"],"source_locator":"README.md: description; MetaPhlAn 4 paper: Building the expanded SGB catalog and Discussion"},{"label":"Known versions","value":"MetaPhlAn 4 paper and 4.2-linked current documentation; database version is a separate reproducibility requirement.","status":"source_checked","source_ids":["evidence-official-72fdaed158e4ff851fbe","evidence-official-37f825d1226e59cbbfbf"],"source_locator":"README.md: description; MetaPhlAn 4 paper: Building the expanded SGB catalog and Discussion"},{"label":"Training data","value":"Reference-derived marker database rather than neural pretraining; record the exact database release.","status":"source_checked","source_ids":["evidence-official-72fdaed158e4ff851fbe","evidence-official-37f825d1226e59cbbfbf"],"source_locator":"README.md: description; MetaPhlAn 4 paper: Building the expanded SGB catalog and Discussion"},{"label":"Training cutoff","value":"Inapplicable to neural pretraining; reference-database and input-data dates must be recorded for each run.","status":"inapplicable","source_ids":["evidence-official-72fdaed158e4ff851fbe","evidence-official-37f825d1226e59cbbfbf"],"source_locator":"README.md: description; MetaPhlAn 4 paper: Building the expanded SGB catalog and Discussion"},{"label":"Context limits","value":"Shotgun reads; no fixed neural token context.","status":"source_checked","source_ids":["evidence-official-72fdaed158e4ff851fbe","evidence-official-37f825d1226e59cbbfbf"],"source_locator":"README.md: description; MetaPhlAn 4 paper: Building the expanded SGB catalog and Discussion"},{"label":"Weights licence","value":"Inapplicable to this procedural method; marker databases have their own provenance and terms.","status":"inapplicable","source_ids":["evidence-official-72fdaed158e4ff851fbe","evidence-official-37f825d1226e59cbbfbf"],"source_locator":"README.md: description; MetaPhlAn 4 paper: Building the expanded SGB catalog and Discussion"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/biobakery/MetaPhlAn","status":"source_checked","source_ids":["evidence-official-72fdaed158e4ff851fbe","evidence-official-37f825d1226e59cbbfbf"],"source_locator":"README.md: description; MetaPhlAn 4 paper: Building the expanded SGB catalog and Discussion"},{"label":"Code licence","value":"MIT","status":"source_checked","source_ids":["evidence-official-1a0775cac85be75449ef"],"source_locator":"license.txt: licence text"}],"strengths":[{"text":"Marker-based profiling can incorporate characterized and previously uncharacterized species groups.","source_ids":["evidence-official-72fdaed158e4ff851fbe","evidence-official-37f825d1226e59cbbfbf"],"source_locator":"README.md: description; MetaPhlAn 4 paper: Building the expanded SGB catalog and Discussion"}],"limitations":[{"text":"Coverage depends on the marker database and habitat. The MetaPhlAn 4 paper identifies remaining gaps for under-studied environmental communities; newer software/databases may have different scope.","source_ids":["evidence-official-72fdaed158e4ff851fbe","evidence-official-37f825d1226e59cbbfbf"],"source_locator":"README.md: description; MetaPhlAn 4 paper: Building the expanded SGB catalog and Discussion"}],"diagram":{"title":"MetaPhlAn workflow","steps":["Metagenomic reads","Marker-gene mapping","Species-group quantification","Relative-abundance profile"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-72fdaed158e4ff851fbe","evidence-official-37f825d1226e59cbbfbf"],"source_locator":"README.md: description; MetaPhlAn 4 paper: Building the expanded SGB catalog and Discussion"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}},"description":"Marker-based microbial profiling","facets":{"areas":["microbiome"]},"id":"discovery-model-metaphlan","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-cami"}],"name":"MetaPhlAn","source_ids":["src-discovery-biobakery-metaphlan"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-cafa"],"entity_level":"method","reported_name":"MMseqs2","version":null,"historical_missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"MMseqs2 searches and clusters large protein and nucleotide sequence collections.","summary_source_ids":["evidence-official-81e51077d7d3352a6de4"],"summary_source_locator":"README.md: opening, Publications, Installation and user-guide pointers","sections":[{"title":"How it works","body":"MMseqs2 searches and clusters large protein and nucleotide sequence collections. Sequence/profile search and clustering software with CPU and selected GPU execution paths. The documented inputs are protein or nucleotide query sequences and a reference database, or sequences to cluster. The output consists of sequence hits, alignments, clusters or configured taxonomic assignments.","source_ids":["evidence-official-81e51077d7d3352a6de4"],"source_locator":"README.md: opening, Publications, Installation and user-guide pointers"},{"title":"Versions and reproducibility","body":"MMseqs2 software release and database build must both be pinned; GPU capability differs by build. Database/query scale and implementation limits, not a neural token window.","source_ids":["evidence-official-81e51077d7d3352a6de4"],"source_locator":"README.md: opening, Publications, Installation and user-guide pointers"}],"facts":[{"label":"Model type","value":"Sequence search and clustering software","status":"source_checked","source_ids":["evidence-official-81e51077d7d3352a6de4"],"source_locator":"README.md: opening, Publications, Installation and user-guide pointers"},{"label":"Architecture","value":"Sequence/profile search and clustering software with CPU and selected GPU execution paths.","status":"source_checked","source_ids":["evidence-official-81e51077d7d3352a6de4"],"source_locator":"README.md: opening, Publications, Installation and user-guide pointers"},{"label":"Inputs","value":"Protein or nucleotide query sequences and a reference database, or sequences to cluster.","status":"source_checked","source_ids":["evidence-official-81e51077d7d3352a6de4"],"source_locator":"README.md: opening, Publications, Installation and user-guide pointers"},{"label":"Outputs","value":"Sequence hits, alignments, clusters or configured taxonomic assignments.","status":"source_checked","source_ids":["evidence-official-81e51077d7d3352a6de4"],"source_locator":"README.md: opening, Publications, Installation and user-guide pointers"},{"label":"Parameters","value":"Inapplicable as a neural parameter count; search and clustering settings apply.","status":"inapplicable","source_ids":["evidence-official-81e51077d7d3352a6de4"],"source_locator":"README.md: opening, Publications, Installation and user-guide pointers"},{"label":"Known versions","value":"MMseqs2 software release and database build must both be pinned; GPU capability differs by build.","status":"source_checked","source_ids":["evidence-official-81e51077d7d3352a6de4"],"source_locator":"README.md: opening, Publications, Installation and user-guide pointers"},{"label":"Training data","value":"Reference sequence/profile databases rather than neural pretraining.","status":"source_checked","source_ids":["evidence-official-81e51077d7d3352a6de4"],"source_locator":"README.md: opening, Publications, Installation and user-guide pointers"},{"label":"Training cutoff","value":"Inapplicable to neural pretraining; reference-database and input-data dates must be recorded for each run.","status":"inapplicable","source_ids":["evidence-official-81e51077d7d3352a6de4"],"source_locator":"README.md: opening, Publications, Installation and user-guide pointers"},{"label":"Context limits","value":"Database/query scale and implementation limits, not a neural token window.","status":"source_checked","source_ids":["evidence-official-81e51077d7d3352a6de4"],"source_locator":"README.md: opening, Publications, Installation and user-guide pointers"},{"label":"Weights licence","value":"Inapplicable: no neural checkpoint in the core search/clustering method.","status":"inapplicable","source_ids":["evidence-official-81e51077d7d3352a6de4"],"source_locator":"README.md: opening, Publications, Installation and user-guide pointers"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/soedinglab/MMseqs2","status":"source_checked","source_ids":["evidence-official-81e51077d7d3352a6de4"],"source_locator":"README.md: opening, Publications, Installation and user-guide pointers"},{"label":"Code licence","value":"MIT","status":"source_checked","source_ids":["evidence-official-950c72bc9d4bc037f6e2"],"source_locator":"LICENSE.md: licence text"}],"strengths":[{"text":"Provides reusable search and clustering procedures suitable as homology-based comparators.","source_ids":["evidence-official-81e51077d7d3352a6de4"],"source_locator":"README.md: opening, Publications, Installation and user-guide pointers"}],"limitations":[{"text":"Hardware paths, sensitivity settings and reference database versions affect an evaluation. A sequence-similarity hit is not itself a validated functional measurement.","source_ids":["evidence-official-81e51077d7d3352a6de4"],"source_locator":"README.md: opening, Publications, Installation and user-guide pointers"}],"diagram":{"title":"MMseqs2 workflow","steps":["Sequence collection","Configured search or clustering","Sequence relationships","Hits or clusters"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-81e51077d7d3352a6de4"],"source_locator":"README.md: opening, Publications, Installation and user-guide pointers"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}},"description":"Sequence search and clustering","facets":{"areas":["protein-function"]},"id":"discovery-model-mmseqs2","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-cafa"}],"name":"MMseqs2","source_ids":["src-discovery-soedinglab-mmseqs2"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-massspecgym"],"entity_level":"family","reported_name":"MSAlign","version":null,"historical_missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"MSAlign retrieves candidate molecules from tandem mass spectra by aligning pretrained molecular and spectral representations.","summary_source_ids":["evidence-official-ba08b650243c9212de3f","evidence-official-760ab2fa8c396aeb796c"],"summary_source_locator":"MSAlign paper Section 3 Architecture and Training, Section 4 on splitting, and Section 5.1 Experimental Setting","sections":[{"title":"How it works","body":"MSAlign retrieves candidate molecules from tandem mass spectra by aligning pretrained molecular and spectral representations. Frozen DreaMS and ChemBERTa encoders connected by lightweight MLP projections trained with a candidate-based contrastive objective. The documented inputs are MS/MS spectrum and a set of candidate molecular structures. The output consists of candidate-molecule retrieval scores in a shared representation space.","source_ids":["evidence-official-ba08b650243c9212de3f","evidence-official-760ab2fa8c396aeb796c"],"source_locator":"MSAlign paper Section 3 Architecture and Training, Section 4 on splitting, and Section 5.1 Experimental Setting"},{"title":"Versions and reproducibility","body":"MSAlign arXiv:2605.19752v1, submitted 19 May 2026. The applicable input limits require configuration-specific checking.","source_ids":["evidence-official-ba08b650243c9212de3f","evidence-official-760ab2fa8c396aeb796c"],"source_locator":"MSAlign paper Section 3 Architecture and Training, Section 4 on splitting, and Section 5.1 Experimental Setting"}],"facts":[{"label":"Model type","value":"Frozen spectral/molecular encoders with learned alignment projections","status":"source_checked","source_ids":["evidence-official-ba08b650243c9212de3f","evidence-official-760ab2fa8c396aeb796c"],"source_locator":"MSAlign paper Section 3 Architecture and Training, Section 4 on splitting, and Section 5.1 Experimental Setting"},{"label":"Architecture","value":"Frozen DreaMS and ChemBERTa encoders connected by lightweight MLP projections trained with a candidate-based contrastive objective.","status":"source_checked","source_ids":["evidence-official-ba08b650243c9212de3f","evidence-official-760ab2fa8c396aeb796c"],"source_locator":"MSAlign paper Section 3 Architecture and Training, Section 4 on splitting, and Section 5.1 Experimental Setting"},{"label":"Inputs","value":"MS/MS spectrum and a set of candidate molecular structures.","status":"source_checked","source_ids":["evidence-official-ba08b650243c9212de3f","evidence-official-760ab2fa8c396aeb796c"],"source_locator":"MSAlign paper Section 3 Architecture and Training, Section 4 on splitting, and Section 5.1 Experimental Setting"},{"label":"Outputs","value":"Candidate-molecule retrieval scores in a shared representation space.","status":"source_checked","source_ids":["evidence-official-ba08b650243c9212de3f","evidence-official-760ab2fa8c396aeb796c"],"source_locator":"MSAlign paper Section 3 Architecture and Training, Section 4 on splitting, and Section 5.1 Experimental Setting"},{"label":"Parameters","value":"Approximately 4M trainable projection parameters; frozen DreaMS and ChemBERTa backbones are reported as 96M and 92M respectively.","status":"source_checked","source_ids":["evidence-official-ba08b650243c9212de3f","evidence-official-760ab2fa8c396aeb796c"],"source_locator":"MSAlign paper Section 3 Architecture and Training, Section 4 on splitting, and Section 5.1 Experimental Setting"},{"label":"Known versions","value":"MSAlign arXiv:2605.19752v1, submitted 19 May 2026.","status":"source_checked","source_ids":["evidence-official-ba08b650243c9212de3f","evidence-official-760ab2fa8c396aeb796c"],"source_locator":"MSAlign paper Section 3 Architecture and Training, Section 4 on splitting, and Section 5.1 Experimental Setting"},{"label":"Training data","value":"Projection layers are fitted separately on the NPLIB1, MassSpecGym or Spectraverse training splits. The paper distinguishes spectrum/molecule pair counts from unique molecules and controls candidate retrieval using mass matching.","status":"source_checked","source_ids":["evidence-official-ba08b650243c9212de3f","evidence-official-760ab2fa8c396aeb796c"],"source_locator":"MSAlign paper Section 3 Architecture and Training, Section 4 on splitting, and Section 5.1 Experimental Setting"},{"label":"Training cutoff","value":"NPLIB1, MassSpecGym and Spectraverse are separately split training resources. The inspected paper does not define one latest measurement date covering all three.","status":"unreported","source_ids":["evidence-official-ba08b650243c9212de3f","evidence-official-760ab2fa8c396aeb796c"],"source_locator":"MSAlign paper Section 3 Architecture and Training, Section 4 on splitting, and Section 5.1 Experimental Setting"},{"label":"Context limits","value":"The pipeline inherits spectrum and molecule preprocessing from its frozen DreaMS and ChemBERTa encoders. The inspected MSAlign architecture section does not state a single joint input limit.","status":"unreported","source_ids":["evidence-official-ba08b650243c9212de3f","evidence-official-760ab2fa8c396aeb796c"],"source_locator":"MSAlign paper Section 3 Architecture and Training, Section 4 on splitting, and Section 5.1 Experimental Setting"},{"label":"Weights licence","value":"The inspected preprint does not state distribution terms for the learned projection checkpoints. Frozen encoder licences remain separate from projection-weight rights.","status":"unreported","source_ids":["evidence-official-ba08b650243c9212de3f","evidence-official-760ab2fa8c396aeb796c"],"source_locator":"MSAlign paper Section 3 Architecture and Training, Section 4 on splitting, and Section 5.1 Experimental Setting; page.html: inspected official source"},{"label":"Access","value":"Official project documentation and implementation: https://arxiv.org/abs/2605.19752","status":"source_checked","source_ids":["evidence-official-ba08b650243c9212de3f","evidence-official-760ab2fa8c396aeb796c"],"source_locator":"MSAlign paper Section 3 Architecture and Training, Section 4 on splitting, and Section 5.1 Experimental Setting"},{"label":"Code licence","value":"The inspected preprint describes the algorithm but does not supply a separate code licence; no code-distribution permission is inferred from the paper licence.","status":"unreported","source_ids":["evidence-official-ba08b650243c9212de3f"],"source_locator":"page.html: inspected official source"}],"strengths":[{"text":"Keeps the large encoders frozen while learning alignment projections for the retrieval task.","source_ids":["evidence-official-ba08b650243c9212de3f","evidence-official-760ab2fa8c396aeb796c"],"source_locator":"MSAlign paper Section 3 Architecture and Training, Section 4 on splitting, and Section 5.1 Experimental Setting"}],"limitations":[{"text":"Candidate construction and splitting change the retrieval problem. The authors explicitly discuss the trade-off between leakage control and distribution shift.","source_ids":["evidence-official-ba08b650243c9212de3f","evidence-official-760ab2fa8c396aeb796c"],"source_locator":"MSAlign paper Section 3 Architecture and Training, Section 4 on splitting, and Section 5.1 Experimental Setting"}],"diagram":{"title":"MSAlign workflow","steps":["Spectrum and candidate molecules","Frozen DreaMS and ChemBERTa","Learned projection alignment","Candidate retrieval"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-ba08b650243c9212de3f","evidence-official-760ab2fa8c396aeb796c"],"source_locator":"MSAlign paper Section 3 Architecture and Training, Section 4 on splitting, and Section 5.1 Experimental Setting"},"coverage":"limited","gaps":["Training cutoff: NPLIB1, MassSpecGym and Spectraverse are separately split training resources. The inspected paper does not define one latest measurement date covering all three.","Context limits: The pipeline inherits spectrum and molecule preprocessing from its frozen DreaMS and ChemBERTa encoders. The inspected MSAlign architecture section does not state a single joint input limit.","Weights licence: The inspected preprint does not state distribution terms for the learned projection checkpoints. Frozen encoder licences remain separate from projection-weight rights.","Code licence: The inspected preprint describes the algorithm but does not supply a separate code licence; no code-distribution permission is inferred from the paper licence."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}},"description":"Spectrum-to-molecule representation alignment","facets":{"areas":["metabolomics"]},"id":"discovery-model-msalign","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-massspecgym"}],"name":"MSAlign","source_ids":["src-discovery-msalign"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-geneb"],"entity_level":"family","reported_name":"Nucleotide Transformer","version":null,"historical_missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"Nucleotide Transformer is a family of DNA encoders pretrained on human or multispecies sequence corpora.","summary_source_ids":["evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-aadbeb0f10ec55d34f5f"],"summary_source_locator":"docs/nucleotide_transformer.md: Model Variants and Sizes, Tokenization and How to use; Nucleotide Transformer paper Methods: Architecture, Pre-training datasets and Training","sections":[{"title":"How it works","body":"Nucleotide Transformer is a family of DNA encoders pretrained on human or multispecies sequence corpora. Encoder-only transformers with 6-mer tokens; v1 uses learned positional encodings and v2 uses rotary positions and SwiGLU. The documented inputs are DNA sequences tokenized into 6-mers, with single-base handling of N and remainder bases. The output consists of contextual representations used in specified downstream prediction workflows.","source_ids":["evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-aadbeb0f10ec55d34f5f"],"source_locator":"docs/nucleotide_transformer.md: Model Variants and Sizes, Tokenization and How to use; Nucleotide Transformer paper Methods: Architecture, Pre-training datasets and Training"},{"title":"Versions and reproducibility","body":"NT-v1 human-reference/1000G/multispecies variants and NT-v2 50M/100M/250M/500M. NT-v3 is a separate architecture described elsewhere in the repository. v1: approximately 6kb; v2: 2,048 tokens, approximately 12kb. Exact base count depends on special and ambiguous tokens.","source_ids":["evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-aadbeb0f10ec55d34f5f"],"source_locator":"docs/nucleotide_transformer.md: Model Variants and Sizes, Tokenization and How to use; Nucleotide Transformer paper Methods: Architecture, Pre-training datasets and Training"}],"facts":[{"label":"Model type","value":"DNA transformer encoder family","status":"source_checked","source_ids":["evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-aadbeb0f10ec55d34f5f"],"source_locator":"docs/nucleotide_transformer.md: Model Variants and Sizes, Tokenization and How to use; Nucleotide Transformer paper Methods: Architecture, Pre-training datasets and Training"},{"label":"Architecture","value":"Encoder-only transformers with 6-mer tokens; v1 uses learned positional encodings and v2 uses rotary positions and SwiGLU.","status":"source_checked","source_ids":["evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-aadbeb0f10ec55d34f5f"],"source_locator":"docs/nucleotide_transformer.md: Model Variants and Sizes, Tokenization and How to use; Nucleotide Transformer paper Methods: Architecture, Pre-training datasets and Training"},{"label":"Inputs","value":"DNA sequences tokenized into 6-mers, with single-base handling of N and remainder bases.","status":"source_checked","source_ids":["evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-aadbeb0f10ec55d34f5f"],"source_locator":"docs/nucleotide_transformer.md: Model Variants and Sizes, Tokenization and How to use; Nucleotide Transformer paper Methods: Architecture, Pre-training datasets and Training"},{"label":"Outputs","value":"Contextual representations used in specified downstream prediction workflows.","status":"source_checked","source_ids":["evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-aadbeb0f10ec55d34f5f"],"source_locator":"docs/nucleotide_transformer.md: Model Variants and Sizes, Tokenization and How to use; Nucleotide Transformer paper Methods: Architecture, Pre-training datasets and Training"},{"label":"Parameters","value":"50M to 2.5B across the documented v1/v2 family.","status":"source_checked","source_ids":["evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-aadbeb0f10ec55d34f5f"],"source_locator":"docs/nucleotide_transformer.md: Model Variants and Sizes, Tokenization and How to use; Nucleotide Transformer paper Methods: Architecture, Pre-training datasets and Training"},{"label":"Known versions","value":"NT-v1 human-reference/1000G/multispecies variants and NT-v2 50M/100M/250M/500M. NT-v3 is a separate architecture described elsewhere in the repository.","status":"source_checked","source_ids":["evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-aadbeb0f10ec55d34f5f"],"source_locator":"docs/nucleotide_transformer.md: Model Variants and Sizes, Tokenization and How to use; Nucleotide Transformer paper Methods: Architecture, Pre-training datasets and Training"},{"label":"Training data","value":"v1 variants use GRCh38, 3,202 human genomes or 850 multispecies genomes; v2 uses the multispecies corpus.","status":"source_checked","source_ids":["evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-aadbeb0f10ec55d34f5f"],"source_locator":"docs/nucleotide_transformer.md: Model Variants and Sizes, Tokenization and How to use; Nucleotide Transformer paper Methods: Architecture, Pre-training datasets and Training"},{"label":"Training cutoff","value":"The paper specifies human-reference, 1000 Genomes and multispecies training collections by variant. A single latest-deposition date for all sequences is not supplied in the inspected pretraining-data section.","status":"unreported","source_ids":["evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-aadbeb0f10ec55d34f5f"],"source_locator":"docs/nucleotide_transformer.md: Model Variants and Sizes, Tokenization and How to use; Nucleotide Transformer paper Methods: Architecture, Pre-training datasets and Training"},{"label":"Context limits","value":"v1: approximately 6kb; v2: 2,048 tokens, approximately 12kb. Exact base count depends on special and ambiguous tokens.","status":"source_checked","source_ids":["evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-aadbeb0f10ec55d34f5f"],"source_locator":"docs/nucleotide_transformer.md: Model Variants and Sizes, Tokenization and How to use; Nucleotide Transformer paper Methods: Architecture, Pre-training datasets and Training"},{"label":"Weights licence","value":"Separate checkpoint-distribution terms are not stated in the inspected release documentation and licence material. The source-code licence alone is not recorded as an explicit weight grant.","status":"unreported","source_ids":["evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-aadbeb0f10ec55d34f5f","evidence-official-7e4b193e47ba209860a1"],"source_locator":"docs/nucleotide_transformer.md: Model Variants and Sizes, Tokenization and How to use; Nucleotide Transformer paper Methods: Architecture, Pre-training datasets and Training; LICENSE.md: licence text"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/instadeepai/nucleotide-transformer","status":"source_checked","source_ids":["evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-aadbeb0f10ec55d34f5f"],"source_locator":"docs/nucleotide_transformer.md: Model Variants and Sizes, Tokenization and How to use; Nucleotide Transformer paper Methods: Architecture, Pre-training datasets and Training"},{"label":"Code licence","value":"CC-BY-NC-SA-4.0","status":"source_checked","source_ids":["evidence-official-7e4b193e47ba209860a1"],"source_locator":"LICENSE.md: licence text"}],"strengths":[{"text":"The family provides multiple sizes and training corpora, allowing those factors to be compared explicitly.","source_ids":["evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-aadbeb0f10ec55d34f5f"],"source_locator":"docs/nucleotide_transformer.md: Model Variants and Sizes, Tokenization and How to use; Nucleotide Transformer paper Methods: Architecture, Pre-training datasets and Training"}],"limitations":[{"text":"Training corpora and architectures differ across v1 and v2. A family-level name is insufficient to reproduce a score.","source_ids":["evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-aadbeb0f10ec55d34f5f"],"source_locator":"docs/nucleotide_transformer.md: Model Variants and Sizes, Tokenization and How to use; Nucleotide Transformer paper Methods: Architecture, Pre-training datasets and Training"}],"diagram":{"title":"Nucleotide Transformer workflow","steps":["DNA sequence","6-mer tokenization","Selected NT encoder","Representation or adapted predictor"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-aadbeb0f10ec55d34f5f"],"source_locator":"docs/nucleotide_transformer.md: Model Variants and Sizes, Tokenization and How to use; Nucleotide Transformer paper Methods: Architecture, Pre-training datasets and Training"},"coverage":"limited","gaps":["Training cutoff: The paper specifies human-reference, 1000 Genomes and multispecies training collections by variant. A single latest-deposition date for all sequences is not supplied in the inspected pretraining-data section.","Weights licence: Separate checkpoint-distribution terms are not stated in the inspected release documentation and licence material. The source-code licence alone is not recorded as an explicit weight grant."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}},"description":"Genomic representation model family","facets":{"areas":["genomics"]},"id":"discovery-model-nucleotide-transformer","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-geneb"}],"name":"Nucleotide Transformer","source_ids":["src-discovery-instadeepai-nucleotide-transformer"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-casp"],"entity_level":"method","reported_name":"OpenFold","version":null,"historical_missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"OpenFold is a trainable PyTorch implementation of AlphaFold 2 and AlphaFold-Multimer workflows.","summary_source_ids":["evidence-official-c0ac4d10941b082e5f86","evidence-official-53934b67953ba33b2c95"],"summary_source_locator":"docs/source/original_readme.md: Features, Inference and Copyright Notice","sections":[{"title":"How it works","body":"OpenFold is a trainable PyTorch implementation of AlphaFold 2 and AlphaFold-Multimer workflows. AlphaFold-compatible folding architecture with configurable attention implementations and a training pipeline. The documented inputs are protein sequence, sequence alignments and optional structural templates, depending on the inference mode. The output consists of predicted protein structures from selected OpenFold or compatible AlphaFold parameters.","source_ids":["evidence-official-c0ac4d10941b082e5f86","evidence-official-53934b67953ba33b2c95"],"source_locator":"docs/source/original_readme.md: Features, Inference and Copyright Notice"},{"title":"Versions and reproducibility","body":"Documented monomer v2.0.1 and multimer v2.3.2 compatibility; distinguish OpenFold and imported AlphaFold parameters. Memory- and configuration-dependent; low-memory attention and CPU offloading are documented.","source_ids":["evidence-official-c0ac4d10941b082e5f86","evidence-official-53934b67953ba33b2c95"],"source_locator":"docs/source/original_readme.md: Features, Inference and Copyright Notice"}],"facts":[{"label":"Model type","value":"Trainable AlphaFold2-compatible structure predictor","status":"source_checked","source_ids":["evidence-official-c0ac4d10941b082e5f86","evidence-official-53934b67953ba33b2c95"],"source_locator":"docs/source/original_readme.md: Features, Inference and Copyright Notice"},{"label":"Architecture","value":"AlphaFold-compatible folding architecture with configurable attention implementations and a training pipeline.","status":"source_checked","source_ids":["evidence-official-c0ac4d10941b082e5f86","evidence-official-53934b67953ba33b2c95"],"source_locator":"docs/source/original_readme.md: Features, Inference and Copyright Notice"},{"label":"Inputs","value":"Protein sequence, sequence alignments and optional structural templates, depending on the inference mode.","status":"source_checked","source_ids":["evidence-official-c0ac4d10941b082e5f86","evidence-official-53934b67953ba33b2c95"],"source_locator":"docs/source/original_readme.md: Features, Inference and Copyright Notice"},{"label":"Outputs","value":"Predicted protein structures from selected OpenFold or compatible AlphaFold parameters.","status":"source_checked","source_ids":["evidence-official-c0ac4d10941b082e5f86","evidence-official-53934b67953ba33b2c95"],"source_locator":"docs/source/original_readme.md: Features, Inference and Copyright Notice"},{"label":"Parameters","value":"The official documentation supports different AlphaFold2-compatible configurations and heads; this family record does not select one checkpoint with one verified total.","status":"unreported","source_ids":["evidence-official-c0ac4d10941b082e5f86","evidence-official-53934b67953ba33b2c95"],"source_locator":"docs/source/original_readme.md: Features, Inference and Copyright Notice"},{"label":"Known versions","value":"Documented monomer v2.0.1 and multimer v2.3.2 compatibility; distinguish OpenFold and imported AlphaFold parameters.","status":"source_checked","source_ids":["evidence-official-c0ac4d10941b082e5f86","evidence-official-53934b67953ba33b2c95"],"source_locator":"docs/source/original_readme.md: Features, Inference and Copyright Notice"},{"label":"Training data","value":"The documentation releases approximately 400,000 MSAs and PDB70 template-hit files; the actual training schedule/checkpoint remains a separate identity.","status":"source_checked","source_ids":["evidence-official-c0ac4d10941b082e5f86","evidence-official-53934b67953ba33b2c95"],"source_locator":"docs/source/original_readme.md: Features, Inference and Copyright Notice"},{"label":"Training cutoff","value":"The inspected implementation documentation does not supply one common cutoff across its supported original, retrained and extended-context model configurations.","status":"unreported","source_ids":["evidence-official-c0ac4d10941b082e5f86","evidence-official-53934b67953ba33b2c95"],"source_locator":"docs/source/original_readme.md: Features, Inference and Copyright Notice"},{"label":"Context limits","value":"Memory- and configuration-dependent; low-memory attention and CPU offloading are documented.","status":"source_checked","source_ids":["evidence-official-c0ac4d10941b082e5f86","evidence-official-53934b67953ba33b2c95"],"source_locator":"docs/source/original_readme.md: Features, Inference and Copyright Notice"},{"label":"Weights licence","value":"The copyright notice specifies CC-BY-4.0 for DeepMind pretrained parameters; this does not automatically establish all OpenFold checkpoint terms.","status":"source_checked","source_ids":["evidence-official-c0ac4d10941b082e5f86","evidence-official-53934b67953ba33b2c95"],"source_locator":"docs/source/original_readme.md: Features, Inference and Copyright Notice"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/aqlaboratory/openfold","status":"source_checked","source_ids":["evidence-official-c0ac4d10941b082e5f86","evidence-official-53934b67953ba33b2c95"],"source_locator":"docs/source/original_readme.md: Features, Inference and Copyright Notice"},{"label":"Code licence","value":"Apache-2.0","status":"source_checked","source_ids":["evidence-official-c7d214a82cfd1afc827b"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"Provides training code, alignment resources and conversion between compatible parameter formats.","source_ids":["evidence-official-c0ac4d10941b082e5f86","evidence-official-53934b67953ba33b2c95"],"source_locator":"docs/source/original_readme.md: Features, Inference and Copyright Notice"}],"limitations":[{"text":"OpenFold-trained weights and DeepMind weights are separate artifacts. The documentation notes deliberate differences, including omitted model ensembling.","source_ids":["evidence-official-c0ac4d10941b082e5f86","evidence-official-53934b67953ba33b2c95"],"source_locator":"docs/source/original_readme.md: Features, Inference and Copyright Notice"}],"diagram":{"title":"OpenFold workflow","steps":["Sequence and alignments","AlphaFold-compatible folding network","Structure module","Predicted protein structure"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-c0ac4d10941b082e5f86","evidence-official-53934b67953ba33b2c95"],"source_locator":"docs/source/original_readme.md: Features, Inference and Copyright Notice"},"coverage":"limited","gaps":["Parameters: The official documentation supports different AlphaFold2-compatible configurations and heads; this family record does not select one checkpoint with one verified total.","Training cutoff: The inspected implementation documentation does not supply one common cutoff across its supported original, retrained and extended-context model configurations."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}},"description":"Trainable protein structure prediction implementation","facets":{"areas":["protein-structure"]},"id":"discovery-model-openfold","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-casp"}],"name":"OpenFold","source_ids":["src-discovery-aqlaboratory-openfold"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":[],"entity_level":"family","reported_name":"Pangolin","version":null,"historical_missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"Pangolin predicts splice-site strength and changes caused by genetic variants.","summary_source_ids":["evidence-official-eaf3efab8850672f44e4","evidence-official-de1ee43bdd6f04de9850"],"summary_source_locator":"Paper: Deep neural network architecture and Generating training and test sets; README.md: Usage","sections":[{"title":"How it works","body":"Pangolin predicts splice-site strength and changes caused by genetic variants. Dilated convolutional network with 16 residual blocks and skip connections; separate probability and usage outputs for heart, liver, brain and testis. The documented inputs are VCF or CSV variants, reference FASTA and matching gene annotations; custom sequence inference is also available. The output consists of predicted increases/decreases in splice-site strength and their positions.","source_ids":["evidence-official-eaf3efab8850672f44e4","evidence-official-de1ee43bdd6f04de9850"],"source_locator":"Paper: Deep neural network architecture and Generating training and test sets; README.md: Usage"},{"title":"Versions and reproducibility","body":"Pangolin implementation; gene-annotation database and selected weights must be recorded with a run. 5,000 bases upstream and downstream each output position; minimum 10,001-base input for one prediction, with 15,000-base training blocks producing 5,000 central outputs.","source_ids":["evidence-official-eaf3efab8850672f44e4","evidence-official-de1ee43bdd6f04de9850"],"source_locator":"Paper: Deep neural network architecture and Generating training and test sets; README.md: Usage"}],"facts":[{"label":"Model type","value":"Dilated convolutional splicing predictor","status":"source_checked","source_ids":["evidence-official-eaf3efab8850672f44e4","evidence-official-de1ee43bdd6f04de9850"],"source_locator":"Paper: Deep neural network architecture and Generating training and test sets; README.md: Usage"},{"label":"Architecture","value":"Dilated convolutional network with 16 residual blocks and skip connections; separate probability and usage outputs for heart, liver, brain and testis.","status":"source_checked","source_ids":["evidence-official-eaf3efab8850672f44e4","evidence-official-de1ee43bdd6f04de9850"],"source_locator":"Paper: Deep neural network architecture and Generating training and test sets; README.md: Usage"},{"label":"Inputs","value":"VCF or CSV variants, reference FASTA and matching gene annotations; custom sequence inference is also available.","status":"source_checked","source_ids":["evidence-official-eaf3efab8850672f44e4","evidence-official-de1ee43bdd6f04de9850"],"source_locator":"Paper: Deep neural network architecture and Generating training and test sets; README.md: Usage"},{"label":"Outputs","value":"Predicted increases/decreases in splice-site strength and their positions.","status":"source_checked","source_ids":["evidence-official-eaf3efab8850672f44e4","evidence-official-de1ee43bdd6f04de9850"],"source_locator":"Paper: Deep neural network architecture and Generating training and test sets; README.md: Usage"},{"label":"Parameters","value":"The reviewed architecture section specifies the dilated residual network, but does not give a complete parameter total for the released ensemble.","status":"unreported","source_ids":["evidence-official-eaf3efab8850672f44e4","evidence-official-de1ee43bdd6f04de9850"],"source_locator":"Paper: Deep neural network architecture and Generating training and test sets; README.md: Usage"},{"label":"Known versions","value":"Pangolin implementation; gene-annotation database and selected weights must be recorded with a run.","status":"source_checked","source_ids":["evidence-official-eaf3efab8850672f44e4","evidence-official-de1ee43bdd6f04de9850"],"source_locator":"Paper: Deep neural network architecture and Generating training and test sets; README.md: Usage"},{"label":"Training data","value":"Human, rhesus macaque, mouse and rat sequence/splicing data. Human test chromosomes 1, 3, 5, 7 and 9 are held out, with homologous training genes filtered using Ensembl BioMart.","status":"source_checked","source_ids":["evidence-official-eaf3efab8850672f44e4","evidence-official-de1ee43bdd6f04de9850"],"source_locator":"Paper: Deep neural network architecture and Generating training and test sets; README.md: Usage"},{"label":"Training cutoff","value":"Training annotations are GENCODE 34 (human), Ensembl 100 (rhesus), GENCODE M25 (mouse) and Ensembl 101 (rat). These component releases do not establish one latest RNA-seq collection date.","status":"source_checked","source_ids":["evidence-official-eaf3efab8850672f44e4","evidence-official-de1ee43bdd6f04de9850"],"source_locator":"Paper: Deep neural network architecture and Generating training and test sets; README.md: Usage"},{"label":"Context limits","value":"5,000 bases upstream and downstream each output position; minimum 10,001-base input for one prediction, with 15,000-base training blocks producing 5,000 central outputs.","status":"source_checked","source_ids":["evidence-official-eaf3efab8850672f44e4","evidence-official-de1ee43bdd6f04de9850"],"source_locator":"Paper: Deep neural network architecture and Generating training and test sets; README.md: Usage"},{"label":"Weights licence","value":"Separate checkpoint-distribution terms are not stated in the inspected release documentation and licence material. The source-code licence alone is not recorded as an explicit weight grant.","status":"unreported","source_ids":["evidence-official-eaf3efab8850672f44e4","evidence-official-de1ee43bdd6f04de9850","evidence-official-dd29c6cbb629171059a6"],"source_locator":"Paper: Deep neural network architecture and Generating training and test sets; README.md: Usage; LICENSE: licence text"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/tkzeng/Pangolin","status":"source_checked","source_ids":["evidence-official-eaf3efab8850672f44e4","evidence-official-de1ee43bdd6f04de9850"],"source_locator":"Paper: Deep neural network architecture and Generating training and test sets; README.md: Usage"},{"label":"Code licence","value":"GPL-3.0; inspect the pinned licence and any file-specific terms.","status":"source_checked","source_ids":["evidence-official-dd29c6cbb629171059a6"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"Supports custom sequences and annotation-aware variant scoring with configurable search distance.","source_ids":["evidence-official-eaf3efab8850672f44e4","evidence-official-de1ee43bdd6f04de9850"],"source_locator":"Paper: Deep neural network architecture and Generating training and test sets; README.md: Usage"}],"limitations":[{"text":"Only substitutions and simple insertions/deletions are supported. The documented tool skips variants outside annotated genes, near chromosome ends, inconsistent with the reference or beyond supported deletion lengths.","source_ids":["evidence-official-eaf3efab8850672f44e4","evidence-official-de1ee43bdd6f04de9850"],"source_locator":"Paper: Deep neural network architecture and Generating training and test sets; README.md: Usage"}],"diagram":{"title":"Pangolin workflow","steps":["Variant and reference genome","Construct sequence inputs","Splice-strength prediction","Reference/alternate comparison"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-eaf3efab8850672f44e4","evidence-official-de1ee43bdd6f04de9850"],"source_locator":"Paper: Deep neural network architecture and Generating training and test sets; README.md: Usage"},"coverage":"limited","gaps":["Parameters: The reviewed architecture section specifies the dilated residual network, but does not give a complete parameter total for the released ensemble.","Weights licence: Separate checkpoint-distribution terms are not stated in the inspected release documentation and licence material. The source-code licence alone is not recorded as an explicit weight grant."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}},"description":"Splice site strength prediction","facets":{"areas":["genomics"]},"id":"discovery-model-pangolin","kind":"model","links":[],"name":"Pangolin","source_ids":["src-discovery-tkzeng-pangolin"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-proteinbench"],"entity_level":"family","reported_name":"ProteinMPNN","version":null,"historical_missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"ProteinMPNN designs amino-acid sequences for a supplied protein backbone.","summary_source_ids":["evidence-official-ea568a52ea3ab476db6a","evidence-official-7b6fdf915d9c6950ad0d","evidence-official-019aff235ae2c3cf29d6","evidence-official-92b0ae57a020cbc40fe7","evidence-official-0b8003691827eb09b0a4"],"summary_source_locator":"ProteinMPNN paper: main-text model development; training/README.md: multi-chain training set, list.csv and cluster manifests; protein_mpnn_utils.py: ProteinMPNN; Supplementary Materials: Methods for training multi-chain models / Training data","sections":[{"title":"How it works","body":"ProteinMPNN converts a supplied backbone into a graph whose edges encode interatomic distances. Message-passing layers update node and edge features, and an autoregressive decoder samples amino acids while conditioning on the backbone and previously assigned residues. Fixed residues, tied positions and chain choices change the design task and must accompany its result.","source_ids":["evidence-official-ea568a52ea3ab476db6a","evidence-official-7b6fdf915d9c6950ad0d","evidence-official-019aff235ae2c3cf29d6","evidence-official-92b0ae57a020cbc40fe7","evidence-official-0b8003691827eb09b0a4"],"source_locator":"ProteinMPNN paper: main-text model development; training/README.md: multi-chain training set, list.csv and cluster manifests; protein_mpnn_utils.py: ProteinMPNN; Supplementary Materials: Methods for training multi-chain models / Training data"},{"title":"Versions and reproducibility","body":"v_48_002, v_48_010, v_48_020 and v_48_030; distinct soluble and C-alpha-only weights. Structure-size and memory dependent. README --max_length is an implementation guard, not a validated scientific context limit.","source_ids":["evidence-official-ea568a52ea3ab476db6a","evidence-official-7b6fdf915d9c6950ad0d","evidence-official-019aff235ae2c3cf29d6","evidence-official-92b0ae57a020cbc40fe7","evidence-official-0b8003691827eb09b0a4"],"source_locator":"ProteinMPNN paper: main-text model development; training/README.md: multi-chain training set, list.csv and cluster manifests; protein_mpnn_utils.py: ProteinMPNN; Supplementary Materials: Methods for training multi-chain models / Training data"}],"facts":[{"label":"Model type","value":"Structure-conditioned message-passing sequence design model","status":"source_checked","source_ids":["evidence-official-ea568a52ea3ab476db6a","evidence-official-7b6fdf915d9c6950ad0d","evidence-official-019aff235ae2c3cf29d6","evidence-official-92b0ae57a020cbc40fe7","evidence-official-0b8003691827eb09b0a4"],"source_locator":"ProteinMPNN paper: main-text model development; training/README.md: multi-chain training set, list.csv and cluster manifests; protein_mpnn_utils.py: ProteinMPNN; Supplementary Materials: Methods for training multi-chain models / Training data"},{"label":"Architecture","value":"Message-passing encoder-decoder with structural interatomic-distance features and edge updates; sequences are sampled with the configured autoregressive decoding procedure.","status":"source_checked","source_ids":["evidence-official-ea568a52ea3ab476db6a","evidence-official-7b6fdf915d9c6950ad0d","evidence-official-019aff235ae2c3cf29d6","evidence-official-92b0ae57a020cbc40fe7","evidence-official-0b8003691827eb09b0a4"],"source_locator":"ProteinMPNN paper: main-text model development; training/README.md: multi-chain training set, list.csv and cluster manifests; protein_mpnn_utils.py: ProteinMPNN; Supplementary Materials: Methods for training multi-chain models / Training data"},{"label":"Inputs","value":"Protein backbone coordinates, with optional fixed residues, chain choices, tied positions and amino-acid constraints.","status":"source_checked","source_ids":["evidence-official-ea568a52ea3ab476db6a","evidence-official-7b6fdf915d9c6950ad0d","evidence-official-019aff235ae2c3cf29d6","evidence-official-92b0ae57a020cbc40fe7","evidence-official-0b8003691827eb09b0a4"],"source_locator":"ProteinMPNN paper: main-text model development; training/README.md: multi-chain training set, list.csv and cluster manifests; protein_mpnn_utils.py: ProteinMPNN; Supplementary Materials: Methods for training multi-chain models / Training data"},{"label":"Outputs","value":"Designed sequences, sequence scores and conditional amino-acid probabilities.","status":"source_checked","source_ids":["evidence-official-ea568a52ea3ab476db6a","evidence-official-7b6fdf915d9c6950ad0d","evidence-official-019aff235ae2c3cf29d6","evidence-official-92b0ae57a020cbc40fe7","evidence-official-0b8003691827eb09b0a4"],"source_locator":"ProteinMPNN paper: main-text model development; training/README.md: multi-chain training set, list.csv and cluster manifests; protein_mpnn_utils.py: ProteinMPNN; Supplementary Materials: Methods for training multi-chain models / Training data"},{"label":"Parameters","value":"The inspected model implementation is configured through encoder/decoder depth and feature width. The paper and training README do not state an exact total for every released checkpoint.","status":"unreported","source_ids":["evidence-official-ea568a52ea3ab476db6a","evidence-official-7b6fdf915d9c6950ad0d","evidence-official-019aff235ae2c3cf29d6","evidence-official-92b0ae57a020cbc40fe7","evidence-official-0b8003691827eb09b0a4"],"source_locator":"ProteinMPNN paper: main-text model development; training/README.md: multi-chain training set, list.csv and cluster manifests; protein_mpnn_utils.py: ProteinMPNN; Supplementary Materials: Methods for training multi-chain models / Training data"},{"label":"Known versions","value":"v_48_002, v_48_010, v_48_020 and v_48_030; distinct soluble and C-alpha-only weights.","status":"source_checked","source_ids":["evidence-official-ea568a52ea3ab476db6a","evidence-official-7b6fdf915d9c6950ad0d","evidence-official-019aff235ae2c3cf29d6","evidence-official-92b0ae57a020cbc40fe7","evidence-official-0b8003691827eb09b0a4"],"source_locator":"ProteinMPNN paper: main-text model development; training/README.md: multi-chain training set, list.csv and cluster manifests; protein_mpnn_utils.py: ProteinMPNN; Supplementary Materials: Methods for training multi-chain models / Training data"},{"label":"Training data","value":"Released multi-chain training set of PDB biological units, with chain metadata and validation/test cluster manifests. The documented set is dated 2 August 2021.","status":"source_checked","source_ids":["evidence-official-ea568a52ea3ab476db6a","evidence-official-7b6fdf915d9c6950ad0d","evidence-official-019aff235ae2c3cf29d6","evidence-official-92b0ae57a020cbc40fe7","evidence-official-0b8003691827eb09b0a4"],"source_locator":"ProteinMPNN paper: main-text model development; training/README.md: multi-chain training set, list.csv and cluster manifests; protein_mpnn_utils.py: ProteinMPNN; Supplementary Materials: Methods for training multi-chain models / Training data"},{"label":"Training cutoff","value":"The released PDB training-set snapshot is dated 2021-08-02; preserve its chain-level deposition metadata and cluster split for a run.","status":"source_checked","source_ids":["evidence-official-ea568a52ea3ab476db6a","evidence-official-7b6fdf915d9c6950ad0d","evidence-official-019aff235ae2c3cf29d6","evidence-official-92b0ae57a020cbc40fe7","evidence-official-0b8003691827eb09b0a4"],"source_locator":"ProteinMPNN paper: main-text model development; training/README.md: multi-chain training set, list.csv and cluster manifests; protein_mpnn_utils.py: ProteinMPNN; Supplementary Materials: Methods for training multi-chain models / Training data"},{"label":"Context limits","value":"Structure-size and memory dependent. README --max_length is an implementation guard, not a validated scientific context limit.","status":"source_checked","source_ids":["evidence-official-ea568a52ea3ab476db6a","evidence-official-7b6fdf915d9c6950ad0d","evidence-official-019aff235ae2c3cf29d6","evidence-official-92b0ae57a020cbc40fe7","evidence-official-0b8003691827eb09b0a4"],"source_locator":"ProteinMPNN paper: main-text model development; training/README.md: multi-chain training set, list.csv and cluster manifests; protein_mpnn_utils.py: ProteinMPNN; Supplementary Materials: Methods for training multi-chain models / Training data"},{"label":"Weights licence","value":"Separate checkpoint-distribution terms are not stated in the inspected release documentation and licence material. The source-code licence alone is not recorded as an explicit weight grant.","status":"unreported","source_ids":["evidence-official-ea568a52ea3ab476db6a","evidence-official-7b6fdf915d9c6950ad0d","evidence-official-019aff235ae2c3cf29d6","evidence-official-92b0ae57a020cbc40fe7","evidence-official-0b8003691827eb09b0a4","evidence-official-eebe7c91156963e6ddc0"],"source_locator":"ProteinMPNN paper: main-text model development; training/README.md: multi-chain training set, list.csv and cluster manifests; protein_mpnn_utils.py: ProteinMPNN; Supplementary Materials: Methods for training multi-chain models / Training data; LICENSE: licence text"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/dauparas/ProteinMPNN","status":"source_checked","source_ids":["evidence-official-ea568a52ea3ab476db6a","evidence-official-7b6fdf915d9c6950ad0d","evidence-official-019aff235ae2c3cf29d6","evidence-official-92b0ae57a020cbc40fe7","evidence-official-0b8003691827eb09b0a4"],"source_locator":"ProteinMPNN paper: main-text model development; training/README.md: multi-chain training set, list.csv and cluster manifests; protein_mpnn_utils.py: ProteinMPNN; Supplementary Materials: Methods for training multi-chain models / Training data"},{"label":"Code licence","value":"MIT","status":"source_checked","source_ids":["evidence-official-eebe7c91156963e6ddc0"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"The interface makes design constraints explicit and includes full-backbone, C-alpha-only and soluble-protein weight sets.","source_ids":["evidence-official-ea568a52ea3ab476db6a","evidence-official-7b6fdf915d9c6950ad0d","evidence-official-019aff235ae2c3cf29d6","evidence-official-92b0ae57a020cbc40fe7","evidence-official-0b8003691827eb09b0a4"],"source_locator":"ProteinMPNN paper: main-text model development; training/README.md: multi-chain training set, list.csv and cluster manifests; protein_mpnn_utils.py: ProteinMPNN; Supplementary Materials: Methods for training multi-chain models / Training data"}],"limitations":[{"text":"The requested backbone and constraints are part of the evaluated problem. A command-line maximum-length guard is not evidence that designs at that size have been validated.","source_ids":["evidence-official-ea568a52ea3ab476db6a","evidence-official-7b6fdf915d9c6950ad0d","evidence-official-019aff235ae2c3cf29d6","evidence-official-92b0ae57a020cbc40fe7","evidence-official-0b8003691827eb09b0a4"],"source_locator":"ProteinMPNN paper: main-text model development; training/README.md: multi-chain training set, list.csv and cluster manifests; protein_mpnn_utils.py: ProteinMPNN; Supplementary Materials: Methods for training multi-chain models / Training data"}],"diagram":{"title":"ProteinMPNN workflow","steps":["Protein backbone","Structural graph features","Message-passing model","Constrained sequence sampling"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-ea568a52ea3ab476db6a","evidence-official-7b6fdf915d9c6950ad0d","evidence-official-019aff235ae2c3cf29d6","evidence-official-92b0ae57a020cbc40fe7","evidence-official-0b8003691827eb09b0a4"],"source_locator":"ProteinMPNN paper: main-text model development; training/README.md: multi-chain training set, list.csv and cluster manifests; protein_mpnn_utils.py: ProteinMPNN; Supplementary Materials: Methods for training multi-chain models / Training data"},"coverage":"limited","gaps":["Parameters: The inspected model implementation is configured through encoder/decoder depth and feature width. The paper and training README do not state an exact total for every released checkpoint.","Weights licence: Separate checkpoint-distribution terms are not stated in the inspected release documentation and licence material. The source-code licence alone is not recorded as an explicit weight grant."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}},"description":"Structure-conditioned protein sequence design","facets":{"areas":["protein-structure"]},"id":"discovery-model-proteinmpnn","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-proteinbench"}],"name":"ProteinMPNN","source_ids":["src-discovery-dauparas-proteinmpnn"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-proteinbench"],"entity_level":"family","reported_name":"RFdiffusion","version":null,"historical_missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"RFdiffusion generates protein structures, optionally conditioned on a motif, target or symmetry constraint.","summary_source_ids":["evidence-official-c7befbb0fbd8620d5c3c","evidence-official-c0351b621280a9c081da","evidence-official-28c1720afecf4240f3f7"],"summary_source_locator":"RFdiffusion paper Main text and Extended Data Figure 1 (pretrained representations); README.md: model variants, contigs and licence; Supplementary Methods Sections 1.4 and 4.1, Table 6","sections":[{"title":"How it works","body":"RFdiffusion generates protein structures, optionally conditioned on a motif, target or symmetry constraint. Diffusion-based protein structure generation with checkpoint-specific conditioning and denoising configuration. The documented inputs are unconditional length specification or structural constraints such as a motif, target and contig map. The output consists of generated protein backbone designs for downstream sequence design and assessment.","source_ids":["evidence-official-c7befbb0fbd8620d5c3c","evidence-official-c0351b621280a9c081da","evidence-official-28c1720afecf4240f3f7"],"source_locator":"RFdiffusion paper Main text and Extended Data Figure 1 (pretrained representations); README.md: model variants, contigs and licence; Supplementary Methods Sections 1.4 and 4.1, Table 6"},{"title":"Versions and reproducibility","body":"Base, active-site, sequence-inpainting and other conditioning-specific checkpoints; preserve the selected weight identity. The RFdiffusion training crop is 384 residues (supplementary Table 6). This crop size is not an inference maximum; contig lengths and conditional task settings remain explicit.","source_ids":["evidence-official-c7befbb0fbd8620d5c3c","evidence-official-c0351b621280a9c081da","evidence-official-28c1720afecf4240f3f7"],"source_locator":"RFdiffusion paper Main text and Extended Data Figure 1 (pretrained representations); README.md: model variants, contigs and licence; Supplementary Methods Sections 1.4 and 4.1, Table 6"}],"facts":[{"label":"Model type","value":"Diffusion-based protein backbone generator","status":"source_checked","source_ids":["evidence-official-c7befbb0fbd8620d5c3c","evidence-official-c0351b621280a9c081da","evidence-official-28c1720afecf4240f3f7"],"source_locator":"RFdiffusion paper Main text and Extended Data Figure 1 (pretrained representations); README.md: model variants, contigs and licence; Supplementary Methods Sections 1.4 and 4.1, Table 6"},{"label":"Architecture","value":"Diffusion-based protein structure generation with checkpoint-specific conditioning and denoising configuration.","status":"source_checked","source_ids":["evidence-official-c7befbb0fbd8620d5c3c","evidence-official-c0351b621280a9c081da","evidence-official-28c1720afecf4240f3f7"],"source_locator":"RFdiffusion paper Main text and Extended Data Figure 1 (pretrained representations); README.md: model variants, contigs and licence; Supplementary Methods Sections 1.4 and 4.1, Table 6"},{"label":"Inputs","value":"Unconditional length specification or structural constraints such as a motif, target and contig map.","status":"source_checked","source_ids":["evidence-official-c7befbb0fbd8620d5c3c","evidence-official-c0351b621280a9c081da","evidence-official-28c1720afecf4240f3f7"],"source_locator":"RFdiffusion paper Main text and Extended Data Figure 1 (pretrained representations); README.md: model variants, contigs and licence; Supplementary Methods Sections 1.4 and 4.1, Table 6"},{"label":"Outputs","value":"Generated protein backbone designs for downstream sequence design and assessment.","status":"source_checked","source_ids":["evidence-official-c7befbb0fbd8620d5c3c","evidence-official-c0351b621280a9c081da","evidence-official-28c1720afecf4240f3f7"],"source_locator":"RFdiffusion paper Main text and Extended Data Figure 1 (pretrained representations); README.md: model variants, contigs and licence; Supplementary Methods Sections 1.4 and 4.1, Table 6"},{"label":"Parameters","value":"The complete supplementary architecture and training sections specify RoseTTAFold-derived modules and training settings but do not state a total for each released conditional checkpoint.","status":"unreported","source_ids":["evidence-official-c7befbb0fbd8620d5c3c","evidence-official-c0351b621280a9c081da","evidence-official-28c1720afecf4240f3f7"],"source_locator":"RFdiffusion paper Main text and Extended Data Figure 1 (pretrained representations); README.md: model variants, contigs and licence; Supplementary Methods Sections 1.4 and 4.1, Table 6"},{"label":"Known versions","value":"Base, active-site, sequence-inpainting and other conditioning-specific checkpoints; preserve the selected weight identity.","status":"source_checked","source_ids":["evidence-official-c7befbb0fbd8620d5c3c","evidence-official-c0351b621280a9c081da","evidence-official-28c1720afecf4240f3f7"],"source_locator":"RFdiffusion paper Main text and Extended Data Figure 1 (pretrained representations); README.md: model variants, contigs and licence; Supplementary Methods Sections 1.4 and 4.1, Table 6"},{"label":"Training data","value":"Fine-tunes pretrained RoseTTAFold to denoise protein backbone structures from the PDB; unconditional and task-conditioned variants are distinct configurations.","status":"source_checked","source_ids":["evidence-official-c7befbb0fbd8620d5c3c","evidence-official-c0351b621280a9c081da","evidence-official-28c1720afecf4240f3f7"],"source_locator":"RFdiffusion paper Main text and Extended Data Figure 1 (pretrained representations); README.md: model variants, contigs and licence; Supplementary Methods Sections 1.4 and 4.1, Table 6"},{"label":"Training cutoff","value":"The supplementary RoseTTAFold pretraining description specifies a 2 August 2021 PDB cutoff and additional AlphaFold2 models. This is pretraining provenance, not a date for every conditional design fine-tune.","status":"source_checked","source_ids":["evidence-official-c7befbb0fbd8620d5c3c","evidence-official-c0351b621280a9c081da","evidence-official-28c1720afecf4240f3f7"],"source_locator":"RFdiffusion paper Main text and Extended Data Figure 1 (pretrained representations); README.md: model variants, contigs and licence; Supplementary Methods Sections 1.4 and 4.1, Table 6"},{"label":"Context limits","value":"The RFdiffusion training crop is 384 residues (supplementary Table 6). This crop size is not an inference maximum; contig lengths and conditional task settings remain explicit.","status":"source_checked","source_ids":["evidence-official-c7befbb0fbd8620d5c3c","evidence-official-c0351b621280a9c081da","evidence-official-28c1720afecf4240f3f7"],"source_locator":"RFdiffusion paper Main text and Extended Data Figure 1 (pretrained representations); README.md: model variants, contigs and licence; Supplementary Methods Sections 1.4 and 4.1, Table 6"},{"label":"Weights licence","value":"BSD licence in the inspected LICENSE explicitly covers both source code and linked downloadable model weights.","status":"source_checked","source_ids":["evidence-official-c7befbb0fbd8620d5c3c","evidence-official-c0351b621280a9c081da","evidence-official-28c1720afecf4240f3f7"],"source_locator":"RFdiffusion paper Main text and Extended Data Figure 1 (pretrained representations); README.md: model variants, contigs and licence; Supplementary Methods Sections 1.4 and 4.1, Table 6"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/RosettaCommons/RFdiffusion","status":"source_checked","source_ids":["evidence-official-c7befbb0fbd8620d5c3c","evidence-official-c0351b621280a9c081da","evidence-official-28c1720afecf4240f3f7"],"source_locator":"RFdiffusion paper Main text and Extended Data Figure 1 (pretrained representations); README.md: model variants, contigs and licence; Supplementary Methods Sections 1.4 and 4.1, Table 6"},{"label":"Code licence","value":"BSD-3-Clause","status":"source_checked","source_ids":["evidence-official-fa91833592f376f6fb51"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"Supports motif scaffolding, symmetry, binder design and partial diffusion through explicit conditioning options.","source_ids":["evidence-official-c7befbb0fbd8620d5c3c","evidence-official-c0351b621280a9c081da","evidence-official-28c1720afecf4240f3f7"],"source_locator":"RFdiffusion paper Main text and Extended Data Figure 1 (pretrained representations); README.md: model variants, contigs and licence; Supplementary Methods Sections 1.4 and 4.1, Table 6"}],"limitations":[{"text":"Different conditioning modes use different trained weights. Backbone generation alone does not establish a functional experimentally validated protein.","source_ids":["evidence-official-c7befbb0fbd8620d5c3c","evidence-official-c0351b621280a9c081da","evidence-official-28c1720afecf4240f3f7"],"source_locator":"RFdiffusion paper Main text and Extended Data Figure 1 (pretrained representations); README.md: model variants, contigs and licence; Supplementary Methods Sections 1.4 and 4.1, Table 6"}],"diagram":{"title":"RFdiffusion workflow","steps":["Length or structural constraints","Diffusion sampling","Generated backbone","Separate sequence design and assessment"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-c7befbb0fbd8620d5c3c","evidence-official-c0351b621280a9c081da","evidence-official-28c1720afecf4240f3f7"],"source_locator":"RFdiffusion paper Main text and Extended Data Figure 1 (pretrained representations); README.md: model variants, contigs and licence; Supplementary Methods Sections 1.4 and 4.1, Table 6"},"coverage":"limited","gaps":["Parameters: The complete supplementary architecture and training sections specify RoseTTAFold-derived modules and training settings but do not state a total for each released conditional checkpoint."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}},"description":"Protein structure generation","facets":{"areas":["protein-structure"]},"id":"discovery-model-rfdiffusion","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-proteinbench"}],"name":"RFdiffusion","source_ids":["src-discovery-rosettacommons-rfdiffusion"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-beacon"],"entity_level":"family","reported_name":"RNA-FM","version":null,"historical_missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"RNA-FM learns contextual representations of RNA nucleotides for downstream RNA analyses.","summary_source_ids":["evidence-official-fd8e332abdf04a75195b","evidence-official-e17e3864c464d981afc7"],"summary_source_locator":"README.md: Quick Start, embedding examples and RNA foundation model comparison table","sections":[{"title":"How it works","body":"RNA-FM converts an RNA sequence into one token per nucleotide. Twelve transformer encoder blocks use self-attention to produce a 640-dimensional representation at each position. During pretraining, the model learns to recover masked nucleotides from their surrounding sequence. The resulting representations can be supplied to a separately specified downstream model; they are not, by themselves, a structure or functional prediction.","source_ids":["evidence-final-model-rna-fm-paper"],"source_locator":"arXiv:2204.00300v5, Methods: ncRNA data collection and preprocessing and RNA foundation model training details (p.22)"},{"title":"Versions and reproducibility","body":"The nucleotide-based rna_fm_t12 and codon-based mrna_fm_t12 interfaces are distinct. The original RNA-FM paper sets a training input-length limit of 1,024 and describes a usable limit of 1,022 nucleotides. That limit should not be assigned to mRNA-FM without checking its separate configuration.","source_ids":["evidence-official-fd8e332abdf04a75195b","evidence-official-e17e3864c464d981afc7","evidence-final-model-rna-fm-paper"],"source_locator":"Official README: Quick Start and RNA-FM/mRNA-FM examples; arXiv:2204.00300v5, Methods: training input length and SARS-CoV-2 genome embedding extraction (pp.22–23)"}],"facts":[{"label":"Model type","value":"Masked-token RNA transformer encoder","status":"source_checked","source_ids":["evidence-official-fd8e332abdf04a75195b","evidence-official-e17e3864c464d981afc7"],"source_locator":"README.md: Quick Start, embedding examples and RNA foundation model comparison table"},{"label":"Architecture","value":"12-layer masked-token transformer encoder with hidden width 640 and 20 attention heads; nucleotide tokens produce contextual representations.","status":"source_checked","source_ids":["evidence-final-model-rna-fm-paper"],"source_locator":"arXiv:2204.00300v5, Methods: ncRNA data collection and preprocessing and RNA foundation model training details (p.22)"},{"label":"Inputs","value":"RNA sequences tokenized at nucleotide resolution.","status":"source_checked","source_ids":["evidence-official-fd8e332abdf04a75195b","evidence-official-e17e3864c464d981afc7"],"source_locator":"README.md: Quick Start, embedding examples and RNA foundation model comparison table"},{"label":"Outputs","value":"Contextual token embeddings for a specified downstream RNA task.","status":"source_checked","source_ids":["evidence-official-fd8e332abdf04a75195b","evidence-official-e17e3864c464d981afc7"],"source_locator":"README.md: Quick Start, embedding examples and RNA foundation model comparison table"},{"label":"Parameters","value":"99M, as printed in the official Foundation Models table.","status":"source_checked","source_ids":["evidence-official-fd8e332abdf04a75195b","evidence-official-e17e3864c464d981afc7"],"source_locator":"README.md: Quick Start, embedding examples and RNA foundation model comparison table"},{"label":"Known versions","value":"rna_fm_t12 and mrna_fm_t12 are separate pretrained interfaces.","status":"source_checked","source_ids":["evidence-official-fd8e332abdf04a75195b","evidence-official-e17e3864c464d981afc7"],"source_locator":"README.md: Quick Start, embedding examples and RNA foundation model comparison table"},{"label":"Training data","value":"23.7 million non-coding RNA sequences collected from RNAcentral. The authors replace T with U and remove identical sequences using CD-HIT-EST at 100% identity, naming the resulting corpus RNAcentral100.","status":"source_checked","source_ids":["evidence-final-model-rna-fm-paper"],"source_locator":"arXiv:2204.00300v5, Methods: ncRNA data collection and preprocessing and RNA foundation model training details (p.22)"},{"label":"Training cutoff","value":"The inspected Methods and official README do not establish an exact dated RNAcentral release. RNAcentral100 is the authors’ processed-corpus label, not a verified release number.","status":"unreported","source_ids":["evidence-final-model-rna-fm-paper","evidence-official-fd8e332abdf04a75195b"],"source_locator":"arXiv:2204.00300v5, Methods: ncRNA data collection and preprocessing (p.22); official README: Foundation Models table"},{"label":"Context limits","value":"The original paper sets a training input-length limit of 1,024 and describes a usable input limit of 1,022 nucleotides. These are the original RNA-FM settings, not a validated limit for later codon-based mRNA-FM checkpoints.","status":"source_checked","source_ids":["evidence-final-model-rna-fm-paper"],"source_locator":"arXiv:2204.00300v5, Methods: RNA foundation model training details (pp.22–23) and RNA-FM application input-limit statement (p.23)"},{"label":"Weights licence","value":"Separate checkpoint-distribution terms are not stated in the inspected release documentation and licence material. The source-code licence alone is not recorded as an explicit weight grant.","status":"unreported","source_ids":["evidence-official-fd8e332abdf04a75195b","evidence-official-e17e3864c464d981afc7","evidence-official-3fde3df73e79e455bd86"],"source_locator":"README.md: Quick Start, embedding examples and RNA foundation model comparison table; LICENSE: licence text"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/ml4bio/RNA-FM","status":"source_checked","source_ids":["evidence-official-fd8e332abdf04a75195b","evidence-official-e17e3864c464d981afc7"],"source_locator":"README.md: Quick Start, embedding examples and RNA foundation model comparison table"},{"label":"Code licence","value":"MIT","status":"source_checked","source_ids":["evidence-official-3fde3df73e79e455bd86"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"The repository exposes embedding extraction and examples for downstream RNA analyses.","source_ids":["evidence-official-fd8e332abdf04a75195b","evidence-official-e17e3864c464d981afc7"],"source_locator":"README.md: Quick Start, embedding examples and RNA foundation model comparison table"}],"limitations":[{"text":"Base-level RNA-FM and codon-level mRNA-FM are not interchangeable. The task head and tokenization must be specified in each evaluation.","source_ids":["evidence-official-fd8e332abdf04a75195b","evidence-official-e17e3864c464d981afc7"],"source_locator":"README.md: Quick Start, embedding examples and RNA foundation model comparison table"}],"diagram":{"title":"RNA-FM workflow","steps":["RNA sequence","Nucleotide tokenizer","12-layer transformer encoder","Contextual nucleotide representations"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-fd8e332abdf04a75195b","evidence-official-e17e3864c464d981afc7"],"source_locator":"README.md: Quick Start, embedding examples and RNA foundation model comparison table"},"coverage":"limited","gaps":["Weights licence: Separate checkpoint-distribution terms are not stated in the inspected release documentation and licence material. The source-code licence alone is not recorded as an explicit weight grant.","Training cutoff: An exact dated RNAcentral release is not established by the inspected original Methods or official README."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied. Follow-up retrieved the complete original RNA-FM PDF and inspected the full pinned Geneformer repository inventory; unavailable labels were updated only where new evidence resolved the earlier retrieval gap. Follow-up audit reconciles the narrative with the verified RNA-FM input limit and clarifies the original paper’s RNAcentral100 preprocessing definition; mRNA-FM remains separate."}}},"description":"RNA sequence representation family","facets":{"areas":["rna"]},"id":"discovery-model-rna-fm","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-beacon"}],"name":"RNA-FM","source_ids":["src-discovery-ml4bio-rna-fm"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-perturbench"],"entity_level":"family","reported_name":"scGPT","version":null,"historical_missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"scGPT learns representations of single-cell molecular measurements and supports task-specific adaptation.","summary_source_ids":["evidence-official-0e4dac85aa0e0ff1d4b2","evidence-official-d8e2d4f2e752370b6565","evidence-official-efa6ec9d886a21534a20"],"summary_source_locator":"May 2023 scGPT preprint Sections 4.1–4.3 and 4.8; current scgpt/model/model.py and README pretrained model table; README.md: Pretrained scGPT Model Zoo, whole-human row","sections":[{"title":"How it works","body":"scGPT combines each gene identity with its expression-value encoding before transformer attention. The implementation supports several expression encoders and cell-pooling choices. Task heads then predict expression or cell labels; optional masking and batch objectives depend on the training configuration.","source_ids":["evidence-official-0e4dac85aa0e0ff1d4b2","evidence-official-d8e2d4f2e752370b6565","evidence-official-efa6ec9d886a21534a20"],"source_locator":"May 2023 scGPT preprint Sections 4.1–4.3 and 4.8; current scgpt/model/model.py and README pretrained model table; README.md: Pretrained scGPT Model Zoo, whole-human row"},{"title":"Versions and reproducibility","body":"The May 2023 preprint reports an early 10M-cell model. The current whole-human checkpoint table reports 33M normal human cells; these sources describe different releases. The applicable input limits require configuration-specific checking.","source_ids":["evidence-official-0e4dac85aa0e0ff1d4b2","evidence-official-d8e2d4f2e752370b6565","evidence-official-efa6ec9d886a21534a20"],"source_locator":"May 2023 scGPT preprint Sections 4.1–4.3 and 4.8; current scgpt/model/model.py and README pretrained model table; README.md: Pretrained scGPT Model Zoo, whole-human row"}],"facts":[{"label":"Model type","value":"Generative single-cell transformer","status":"source_checked","source_ids":["evidence-official-0e4dac85aa0e0ff1d4b2","evidence-official-d8e2d4f2e752370b6565","evidence-official-efa6ec9d886a21534a20"],"source_locator":"May 2023 scGPT preprint Sections 4.1–4.3 and 4.8; current scgpt/model/model.py and README pretrained model table; README.md: Pretrained scGPT Model Zoo, whole-human row"},{"label":"Architecture","value":"Transformer backbone combining learned gene-token embeddings with expression-value encodings and optional batch encodings. Separate expression, cell-classification and optional masked-value or batch-discriminator heads support configured tasks.","status":"source_checked","source_ids":["evidence-official-0e4dac85aa0e0ff1d4b2","evidence-official-d8e2d4f2e752370b6565","evidence-official-efa6ec9d886a21534a20"],"source_locator":"May 2023 scGPT preprint Sections 4.1–4.3 and 4.8; current scgpt/model/model.py and README pretrained model table; README.md: Pretrained scGPT Model Zoo, whole-human row"},{"label":"Inputs","value":"Gene-expression measurements with the checkpoint-matched gene vocabulary.","status":"source_checked","source_ids":["evidence-official-0e4dac85aa0e0ff1d4b2","evidence-official-d8e2d4f2e752370b6565","evidence-official-efa6ec9d886a21534a20"],"source_locator":"May 2023 scGPT preprint Sections 4.1–4.3 and 4.8; current scgpt/model/model.py and README pretrained model table; README.md: Pretrained scGPT Model Zoo, whole-human row"},{"label":"Outputs","value":"Cell/gene representations and task-specific predictions after the relevant workflow.","status":"source_checked","source_ids":["evidence-official-0e4dac85aa0e0ff1d4b2","evidence-official-d8e2d4f2e752370b6565","evidence-official-efa6ec9d886a21534a20"],"source_locator":"May 2023 scGPT preprint Sections 4.1–4.3 and 4.8; current scgpt/model/model.py and README pretrained model table; README.md: Pretrained scGPT Model Zoo, whole-human row"},{"label":"Parameters","value":"The May 2023 model has 12 transformer blocks, width 512 and eight heads. The inspected current model-zoo table does not state the exact parameter total of its separate 33M-cell checkpoint.","status":"unreported","source_ids":["evidence-official-0e4dac85aa0e0ff1d4b2","evidence-official-d8e2d4f2e752370b6565","evidence-official-efa6ec9d886a21534a20"],"source_locator":"May 2023 scGPT preprint Sections 4.1–4.3 and 4.8; current scgpt/model/model.py and README pretrained model table; README.md: Pretrained scGPT Model Zoo, whole-human row"},{"label":"Known versions","value":"The May 2023 preprint reports an early 10M-cell model. The current whole-human checkpoint table reports 33M normal human cells; these sources describe different releases.","status":"source_checked","source_ids":["evidence-official-0e4dac85aa0e0ff1d4b2","evidence-official-d8e2d4f2e752370b6565","evidence-official-efa6ec9d886a21534a20"],"source_locator":"May 2023 scGPT preprint Sections 4.1–4.3 and 4.8; current scgpt/model/model.py and README pretrained model table; README.md: Pretrained scGPT Model Zoo, whole-human row"},{"label":"Training data","value":"The current whole-human model-zoo checkpoint uses 33M normal human cells, alongside separately released organ-specific and pan-cancer models. The earlier May 2023 preprint describes 10M training cells; its corpus is not the current checkpoint corpus.","status":"source_checked","source_ids":["evidence-official-0e4dac85aa0e0ff1d4b2","evidence-official-d8e2d4f2e752370b6565","evidence-official-efa6ec9d886a21534a20"],"source_locator":"May 2023 scGPT preprint Sections 4.1–4.3 and 4.8; current scgpt/model/model.py and README pretrained model table; README.md: Pretrained scGPT Model Zoo, whole-human row"},{"label":"Training cutoff","value":"The reviewed early manuscript and current whole-human model-zoo entry describe different corpora; neither supplies a shared latest-study date for the current checkpoint.","status":"unreported","source_ids":["evidence-official-0e4dac85aa0e0ff1d4b2","evidence-official-d8e2d4f2e752370b6565","evidence-official-efa6ec9d886a21534a20"],"source_locator":"May 2023 scGPT preprint Sections 4.1–4.3 and 4.8; current scgpt/model/model.py and README pretrained model table; README.md: Pretrained scGPT Model Zoo, whole-human row"},{"label":"Context limits","value":"The implementation accepts variable gene sets matched to its vocabulary. The reviewed model-zoo entry does not specify one validated maximum gene sequence for the current whole-human checkpoint.","status":"unreported","source_ids":["evidence-official-0e4dac85aa0e0ff1d4b2","evidence-official-d8e2d4f2e752370b6565","evidence-official-efa6ec9d886a21534a20"],"source_locator":"May 2023 scGPT preprint Sections 4.1–4.3 and 4.8; current scgpt/model/model.py and README pretrained model table; README.md: Pretrained scGPT Model Zoo, whole-human row"},{"label":"Weights licence","value":"Separate checkpoint-distribution terms are not stated in the inspected release documentation and licence material. The source-code licence alone is not recorded as an explicit weight grant.","status":"unreported","source_ids":["evidence-official-0e4dac85aa0e0ff1d4b2","evidence-official-d8e2d4f2e752370b6565","evidence-official-efa6ec9d886a21534a20","evidence-official-43484d4de29aacd65ed7"],"source_locator":"May 2023 scGPT preprint Sections 4.1–4.3 and 4.8; current scgpt/model/model.py and README pretrained model table; README.md: Pretrained scGPT Model Zoo, whole-human row; LICENSE: licence text"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/bowang-lab/scGPT","status":"source_checked","source_ids":["evidence-official-0e4dac85aa0e0ff1d4b2","evidence-official-d8e2d4f2e752370b6565","evidence-official-efa6ec9d886a21534a20"],"source_locator":"May 2023 scGPT preprint Sections 4.1–4.3 and 4.8; current scgpt/model/model.py and README pretrained model table; README.md: Pretrained scGPT Model Zoo, whole-human row"},{"label":"Code licence","value":"MIT","status":"source_checked","source_ids":["evidence-official-43484d4de29aacd65ed7"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"Releases include whole-human and organ-specific checkpoints, with tutorials for reference mapping and other downstream tasks.","source_ids":["evidence-official-0e4dac85aa0e0ff1d4b2","evidence-official-d8e2d4f2e752370b6565","evidence-official-efa6ec9d886a21534a20"],"source_locator":"May 2023 scGPT preprint Sections 4.1–4.3 and 4.8; current scgpt/model/model.py and README pretrained model table; README.md: Pretrained scGPT Model Zoo, whole-human row"}],"limitations":[{"text":"Checkpoint choice and vocabulary must match the biological context. A whole-human pretrained encoder and a fine-tuned annotation or perturbation model are distinct evaluated configurations.","source_ids":["evidence-official-0e4dac85aa0e0ff1d4b2","evidence-official-d8e2d4f2e752370b6565","evidence-official-efa6ec9d886a21534a20"],"source_locator":"May 2023 scGPT preprint Sections 4.1–4.3 and 4.8; current scgpt/model/model.py and README pretrained model table; README.md: Pretrained scGPT Model Zoo, whole-human row"}],"diagram":{"title":"scGPT workflow","steps":["Gene expression and vocabulary","scGPT encoder","Cell and gene representations","Task-specific adaptation"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-0e4dac85aa0e0ff1d4b2","evidence-official-d8e2d4f2e752370b6565","evidence-official-efa6ec9d886a21534a20"],"source_locator":"May 2023 scGPT preprint Sections 4.1–4.3 and 4.8; current scgpt/model/model.py and README pretrained model table; README.md: Pretrained scGPT Model Zoo, whole-human row"},"coverage":"limited","gaps":["Parameters: The May 2023 model has 12 transformer blocks, width 512 and eight heads. The inspected current model-zoo table does not state the exact parameter total of its separate 33M-cell checkpoint.","Training cutoff: The reviewed early manuscript and current whole-human model-zoo entry describe different corpora; neither supplies a shared latest-study date for the current checkpoint.","Context limits: The implementation accepts variable gene sets matched to its vocabulary. The reviewed model-zoo entry does not specify one validated maximum gene sequence for the current whole-human checkpoint.","Weights licence: Separate checkpoint-distribution terms are not stated in the inspected release documentation and licence material. The source-code licence alone is not recorded as an explicit weight grant."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}},"description":"Single-cell multi-omics model","facets":{"areas":["single-cell"]},"id":"discovery-model-scgpt","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-perturbench"}],"name":"scGPT","source_ids":["src-discovery-bowang-lab-scgpt"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-scib"],"entity_level":"family","reported_name":"scVI","version":null,"historical_missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"scVI models single-cell RNA counts with a probabilistic latent-variable model that accounts for observed covariates.","summary_source_ids":["evidence-official-550f6719c2f22f4d57b3","evidence-official-0cbab80c2d1769287a1f"],"summary_source_locator":"docs/user_guide/models/scvi.md: Preliminaries, Generative process, Inference and Limitations","sections":[{"title":"How it works","body":"scVI models single-cell RNA counts with a probabilistic latent-variable model that accounts for observed covariates. Variational autoencoder with a count likelihood and neural encoder/decoder; likelihood and batch/dispersion settings are configurable. The documented inputs are cell-by-gene count matrix, optionally with batch, donor or other covariates. The output consists of low-dimensional cell representations, normalized expression and probabilistic downstream quantities.","source_ids":["evidence-official-550f6719c2f22f4d57b3","evidence-official-0cbab80c2d1769287a1f"],"source_locator":"docs/user_guide/models/scvi.md: Preliminaries, Generative process, Inference and Limitations"},{"title":"Versions and reproducibility","body":"scVI model within scvi-tools; package version, likelihood, covariates and checkpoint are evaluation-specific. Gene-feature matrix rather than a fixed sequence-token window.","source_ids":["evidence-official-550f6719c2f22f4d57b3","evidence-official-0cbab80c2d1769287a1f"],"source_locator":"docs/user_guide/models/scvi.md: Preliminaries, Generative process, Inference and Limitations"}],"facts":[{"label":"Model type","value":"Variational autoencoder for count data","status":"source_checked","source_ids":["evidence-official-550f6719c2f22f4d57b3","evidence-official-0cbab80c2d1769287a1f"],"source_locator":"docs/user_guide/models/scvi.md: Preliminaries, Generative process, Inference and Limitations"},{"label":"Architecture","value":"Variational autoencoder with a count likelihood and neural encoder/decoder; likelihood and batch/dispersion settings are configurable.","status":"source_checked","source_ids":["evidence-official-550f6719c2f22f4d57b3","evidence-official-0cbab80c2d1769287a1f"],"source_locator":"docs/user_guide/models/scvi.md: Preliminaries, Generative process, Inference and Limitations"},{"label":"Inputs","value":"Cell-by-gene count matrix, optionally with batch, donor or other covariates.","status":"source_checked","source_ids":["evidence-official-550f6719c2f22f4d57b3","evidence-official-0cbab80c2d1769287a1f"],"source_locator":"docs/user_guide/models/scvi.md: Preliminaries, Generative process, Inference and Limitations"},{"label":"Outputs","value":"Low-dimensional cell representations, normalized expression and probabilistic downstream quantities.","status":"source_checked","source_ids":["evidence-official-550f6719c2f22f4d57b3","evidence-official-0cbab80c2d1769287a1f"],"source_locator":"docs/user_guide/models/scvi.md: Preliminaries, Generative process, Inference and Limitations"},{"label":"Parameters","value":"Configuration-dependent, including gene count and encoder/decoder dimensions.","status":"source_checked","source_ids":["evidence-official-550f6719c2f22f4d57b3","evidence-official-0cbab80c2d1769287a1f"],"source_locator":"docs/user_guide/models/scvi.md: Preliminaries, Generative process, Inference and Limitations"},{"label":"Known versions","value":"scVI model within scvi-tools; package version, likelihood, covariates and checkpoint are evaluation-specific.","status":"source_checked","source_ids":["evidence-official-550f6719c2f22f4d57b3","evidence-official-0cbab80c2d1769287a1f"],"source_locator":"docs/user_guide/models/scvi.md: Preliminaries, Generative process, Inference and Limitations"},{"label":"Training data","value":"Fitted to the user-selected count matrix or a specified pretrained reference; scVI is not one universal checkpoint.","status":"source_checked","source_ids":["evidence-official-550f6719c2f22f4d57b3","evidence-official-0cbab80c2d1769287a1f"],"source_locator":"docs/user_guide/models/scvi.md: Preliminaries, Generative process, Inference and Limitations"},{"label":"Training cutoff","value":"Inapplicable as one universal pretraining date: scVI is fitted to the supplied dataset, whose collection date and train/test split belong to the evaluation.","status":"inapplicable","source_ids":["evidence-official-550f6719c2f22f4d57b3","evidence-official-0cbab80c2d1769287a1f"],"source_locator":"docs/user_guide/models/scvi.md: Preliminaries, Generative process, Inference and Limitations"},{"label":"Context limits","value":"Gene-feature matrix rather than a fixed sequence-token window.","status":"source_checked","source_ids":["evidence-official-550f6719c2f22f4d57b3","evidence-official-0cbab80c2d1769287a1f"],"source_locator":"docs/user_guide/models/scvi.md: Preliminaries, Generative process, Inference and Limitations"},{"label":"Weights licence","value":"No universal weights release applies to a model fitted on each dataset; any reused checkpoint requires its own licence.","status":"source_checked","source_ids":["evidence-official-550f6719c2f22f4d57b3","evidence-official-0cbab80c2d1769287a1f"],"source_locator":"docs/user_guide/models/scvi.md: Preliminaries, Generative process, Inference and Limitations"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/scverse/scvi-tools","status":"source_checked","source_ids":["evidence-official-550f6719c2f22f4d57b3","evidence-official-0cbab80c2d1769287a1f"],"source_locator":"docs/user_guide/models/scvi.md: Preliminaries, Generative process, Inference and Limitations"},{"label":"Code licence","value":"BSD-3-Clause","status":"source_checked","source_ids":["evidence-official-6851724e3bcb7e9d2781"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"Models count observations directly and supports batch-conditioned expression estimates and reference-to-query transfer.","source_ids":["evidence-official-550f6719c2f22f4d57b3","evidence-official-0cbab80c2d1769287a1f"],"source_locator":"docs/user_guide/models/scvi.md: Preliminaries, Generative process, Inference and Limitations"}],"limitations":[{"text":"The documentation notes that the latent space is less interpretable than a linear method and efficient inference generally benefits from a GPU. Covariates and likelihood must be reported.","source_ids":["evidence-official-550f6719c2f22f4d57b3","evidence-official-0cbab80c2d1769287a1f"],"source_locator":"docs/user_guide/models/scvi.md: Preliminaries, Generative process, Inference and Limitations"}],"diagram":{"title":"scVI workflow","steps":["RNA counts and covariates","Variational encoder","Latent cell state","Count decoder and estimates"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-550f6719c2f22f4d57b3","evidence-official-0cbab80c2d1769287a1f"],"source_locator":"docs/user_guide/models/scvi.md: Preliminaries, Generative process, Inference and Limitations"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}},"description":"Probabilistic single-cell expression model","facets":{"areas":["single-cell"]},"id":"discovery-model-scvi","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-scib"}],"name":"scVI","source_ids":["src-discovery-scverse-scvi-tools"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":[],"entity_level":"family","reported_name":"SegmentNT","version":null,"historical_missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"SegmentNT labels genomic elements at individual nucleotide positions using a pretrained DNA backbone.","summary_source_ids":["evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-7e59c59bfea79722fc33","evidence-official-97071500fc3422c426d4","evidence-official-da4566a88ed9fa42fdcb"],"summary_source_locator":"SegmentNT paper Methods: SegmentNT architecture, Model ablations, Human genomic elements and Multispecies training; official model-card Training data and licence metadata","sections":[{"title":"How it works","body":"SegmentNT labels genomic elements at individual nucleotide positions using a pretrained DNA backbone. Nucleotide Transformer backbone with a one-dimensional U-Net segmentation head; YaRN rescales positions for longer inputs. The documented inputs are DNA sequences without N bases, tokenized into 6-mers under the documented length constraints. The output consists of per-base probabilities for 14 genomic-element classes.","source_ids":["evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-7e59c59bfea79722fc33","evidence-official-97071500fc3422c426d4","evidence-official-da4566a88ed9fa42fdcb"],"source_locator":"SegmentNT paper Methods: SegmentNT architecture, Model ablations, Human genomic elements and Multispecies training; official model-card Training data and licence metadata"},{"title":"Versions and reproducibility","body":"segment_nt and segment_nt_multi_species; SegmentEnformer and SegmentBorzoi are distinct pipelines. Trained on 30kb; inference up to 50kb requires the documented rescaling. Input token count also has divisibility constraints.","source_ids":["evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-7e59c59bfea79722fc33","evidence-official-97071500fc3422c426d4","evidence-official-da4566a88ed9fa42fdcb"],"source_locator":"SegmentNT paper Methods: SegmentNT architecture, Model ablations, Human genomic elements and Multispecies training; official model-card Training data and licence metadata"}],"facts":[{"label":"Model type","value":"DNA encoder with nucleotide-level segmentation head","status":"source_checked","source_ids":["evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-7e59c59bfea79722fc33","evidence-official-97071500fc3422c426d4","evidence-official-da4566a88ed9fa42fdcb"],"source_locator":"SegmentNT paper Methods: SegmentNT architecture, Model ablations, Human genomic elements and Multispecies training; official model-card Training data and licence metadata"},{"label":"Architecture","value":"Nucleotide Transformer backbone with a one-dimensional U-Net segmentation head; YaRN rescales positions for longer inputs.","status":"source_checked","source_ids":["evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-7e59c59bfea79722fc33","evidence-official-97071500fc3422c426d4","evidence-official-da4566a88ed9fa42fdcb"],"source_locator":"SegmentNT paper Methods: SegmentNT architecture, Model ablations, Human genomic elements and Multispecies training; official model-card Training data and licence metadata"},{"label":"Inputs","value":"DNA sequences without N bases, tokenized into 6-mers under the documented length constraints.","status":"source_checked","source_ids":["evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-7e59c59bfea79722fc33","evidence-official-97071500fc3422c426d4","evidence-official-da4566a88ed9fa42fdcb"],"source_locator":"SegmentNT paper Methods: SegmentNT architecture, Model ablations, Human genomic elements and Multispecies training; official model-card Training data and licence metadata"},{"label":"Outputs","value":"Per-base probabilities for 14 genomic-element classes.","status":"source_checked","source_ids":["evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-7e59c59bfea79722fc33","evidence-official-97071500fc3422c426d4","evidence-official-da4566a88ed9fa42fdcb"],"source_locator":"SegmentNT paper Methods: SegmentNT architecture, Model ablations, Human genomic elements and Multispecies training; official model-card Training data and licence metadata"},{"label":"Parameters","value":"563M for NT-v2 500M plus the 63M segmentation head; alternative encoder ablations are different complete pipelines.","status":"source_checked","source_ids":["evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-7e59c59bfea79722fc33","evidence-official-97071500fc3422c426d4","evidence-official-da4566a88ed9fa42fdcb"],"source_locator":"SegmentNT paper Methods: SegmentNT architecture, Model ablations, Human genomic elements and Multispecies training; official model-card Training data and licence metadata"},{"label":"Known versions","value":"segment_nt and segment_nt_multi_species; SegmentEnformer and SegmentBorzoi are distinct pipelines.","status":"source_checked","source_ids":["evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-7e59c59bfea79722fc33","evidence-official-97071500fc3422c426d4","evidence-official-da4566a88ed9fa42fdcb"],"source_locator":"SegmentNT paper Methods: SegmentNT architecture, Model ablations, Human genomic elements and Multispecies training; official model-card Training data and licence metadata"},{"label":"Training data","value":"Human 14-element segmentation labels from GENCODE V44 and ENCODE SCREEN/DHS annotations. Human chromosomes 20 and 21 are held out for testing and 22 for validation. A separate multispecies model adds mouse, chicken, fly, zebrafish and worm.","status":"source_checked","source_ids":["evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-7e59c59bfea79722fc33","evidence-official-97071500fc3422c426d4","evidence-official-da4566a88ed9fa42fdcb"],"source_locator":"SegmentNT paper Methods: SegmentNT architecture, Model ablations, Human genomic elements and Multispecies training; official model-card Training data and licence metadata"},{"label":"Training cutoff","value":"GENCODE V44 and the named ENCODE SCREEN/DHS resources define label provenance. The inspected paper does not give one latest-experiment date shared by every resource.","status":"source_checked","source_ids":["evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-7e59c59bfea79722fc33","evidence-official-97071500fc3422c426d4","evidence-official-da4566a88ed9fa42fdcb"],"source_locator":"SegmentNT paper Methods: SegmentNT architecture, Model ablations, Human genomic elements and Multispecies training; official model-card Training data and licence metadata"},{"label":"Context limits","value":"Trained on 30kb; inference up to 50kb requires the documented rescaling. Input token count also has divisibility constraints.","status":"source_checked","source_ids":["evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-7e59c59bfea79722fc33","evidence-official-97071500fc3422c426d4","evidence-official-da4566a88ed9fa42fdcb"],"source_locator":"SegmentNT paper Methods: SegmentNT architecture, Model ablations, Human genomic elements and Multispecies training; official model-card Training data and licence metadata"},{"label":"Weights licence","value":"CC-BY-NC-SA-4.0 declared by the official InstaDeepAI/segment_nt model card.","status":"source_checked","source_ids":["evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-7e59c59bfea79722fc33","evidence-official-97071500fc3422c426d4","evidence-official-da4566a88ed9fa42fdcb"],"source_locator":"SegmentNT paper Methods: SegmentNT architecture, Model ablations, Human genomic elements and Multispecies training; official model-card Training data and licence metadata"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/instadeepai/nucleotide-transformer","status":"source_checked","source_ids":["evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-7e59c59bfea79722fc33","evidence-official-97071500fc3422c426d4","evidence-official-da4566a88ed9fa42fdcb"],"source_locator":"SegmentNT paper Methods: SegmentNT architecture, Model ablations, Human genomic elements and Multispecies training; official model-card Training data and licence metadata"},{"label":"Code licence","value":"CC-BY-NC-SA-4.0","status":"source_checked","source_ids":["evidence-official-7e4b193e47ba209860a1"],"source_locator":"LICENSE.md: licence text"}],"strengths":[{"text":"Links sequence representation learning to explicitly localized annotations such as exons, splice sites and regulatory elements.","source_ids":["evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-7e59c59bfea79722fc33","evidence-official-97071500fc3422c426d4","evidence-official-da4566a88ed9fa42fdcb"],"source_locator":"SegmentNT paper Methods: SegmentNT architecture, Model ablations, Human genomic elements and Multispecies training; official model-card Training data and licence metadata"}],"limitations":[{"text":"The documented implementation does not accept N bases. Its 30kb training context and reported 50kb generalization should not be treated as unlimited context.","source_ids":["evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-7e59c59bfea79722fc33","evidence-official-97071500fc3422c426d4","evidence-official-da4566a88ed9fa42fdcb"],"source_locator":"SegmentNT paper Methods: SegmentNT architecture, Model ablations, Human genomic elements and Multispecies training; official model-card Training data and licence metadata"}],"diagram":{"title":"SegmentNT workflow","steps":["DNA sequence","NT backbone with position rescaling","1D U-Net head","Per-base element predictions"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-18d8f4d5fc3f6616922d","evidence-official-657e83427ab59f3aec83","evidence-official-56f02d45976d011d80aa","evidence-official-4920952f9b3c4b8909a0","evidence-official-7e59c59bfea79722fc33","evidence-official-97071500fc3422c426d4","evidence-official-da4566a88ed9fa42fdcb"],"source_locator":"SegmentNT paper Methods: SegmentNT architecture, Model ablations, Human genomic elements and Multispecies training; official model-card Training data and licence metadata"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}},"description":"Genomic sequence segmentation model","facets":{"areas":["genomics"]},"id":"discovery-model-segmentnt","kind":"model","links":[],"name":"SegmentNT","source_ids":["src-discovery-instadeepai-nucleotide-transformer"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":[],"entity_level":"family","reported_name":"SpliceAI","version":null,"historical_missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"SpliceAI annotates sequence variants with predicted splice acceptor and donor changes.","summary_source_ids":["evidence-official-9b820532ba8e3965f64e","evidence-official-ded281404bd1a0f3fdb7"],"summary_source_locator":"Jaganathan et al., Cell 2019, STAR Methods: SpliceAI architecture and Model training and testing (pp. e2–e3); README.md: usage and License","sections":[{"title":"How it works","body":"SpliceAI reads one-hot-encoded DNA through dilated convolutional residual blocks. Skip connections combine features at different depths, and a softmax layer assigns acceptor, donor or neither probabilities to the central positions. The 10kb version requires 5kb of sequence on each side of a scored position; variant scoring compares the reference and alternate predictions.","source_ids":["evidence-official-9b820532ba8e3965f64e","evidence-official-ded281404bd1a0f3fdb7"],"source_locator":"Jaganathan et al., Cell 2019, STAR Methods: SpliceAI architecture and Model training and testing (pp. e2–e3); README.md: usage and License"},{"title":"Versions and reproducibility","body":"The paper studies 80nt, 400nt, 2kb and 10kb receptive spans. Variant scoring averages five independently trained models; these are not five different assay results. SpliceAI-10k uses 5,000 flanking bases on each side. An input of length l + 10,000 produces predictions for l central positions; receptive span is distinct from maximum input length.","source_ids":["evidence-official-9b820532ba8e3965f64e","evidence-official-ded281404bd1a0f3fdb7"],"source_locator":"Jaganathan et al., Cell 2019, STAR Methods: SpliceAI architecture and Model training and testing (pp. e2–e3); README.md: usage and License"}],"facts":[{"label":"Model type","value":"Dilated convolutional splicing predictor","status":"source_checked","source_ids":["evidence-official-9b820532ba8e3965f64e","evidence-official-ded281404bd1a0f3fdb7"],"source_locator":"Jaganathan et al., Cell 2019, STAR Methods: SpliceAI architecture and Model training and testing (pp. e2–e3); README.md: usage and License"},{"label":"Architecture","value":"Residual one-dimensional convolutional network with dilated kernels and skip connections; a softmax head predicts acceptor, donor and neither at each central position.","status":"source_checked","source_ids":["evidence-official-9b820532ba8e3965f64e","evidence-official-ded281404bd1a0f3fdb7"],"source_locator":"Jaganathan et al., Cell 2019, STAR Methods: SpliceAI architecture and Model training and testing (pp. e2–e3); README.md: usage and License"},{"label":"Inputs","value":"VCF variants, reference FASTA and matching gene annotation, or custom one-hot-encoded sequence.","status":"source_checked","source_ids":["evidence-official-9b820532ba8e3965f64e","evidence-official-ded281404bd1a0f3fdb7"],"source_locator":"Jaganathan et al., Cell 2019, STAR Methods: SpliceAI architecture and Model training and testing (pp. e2–e3); README.md: usage and License"},{"label":"Outputs","value":"Acceptor/donor gain/loss scores and positions in VCF INFO annotations.","status":"source_checked","source_ids":["evidence-official-9b820532ba8e3965f64e","evidence-official-ded281404bd1a0f3fdb7"],"source_locator":"Jaganathan et al., Cell 2019, STAR Methods: SpliceAI architecture and Model training and testing (pp. e2–e3); README.md: usage and License"},{"label":"Parameters","value":"The original STAR Methods specifies residual blocks, dilation and receptive spans, but does not state a complete parameter count for each released five-model scoring ensemble.","status":"unreported","source_ids":["evidence-official-9b820532ba8e3965f64e","evidence-official-ded281404bd1a0f3fdb7"],"source_locator":"Jaganathan et al., Cell 2019, STAR Methods: SpliceAI architecture and Model training and testing (pp. e2–e3); README.md: usage and License"},{"label":"Known versions","value":"The paper studies 80nt, 400nt, 2kb and 10kb receptive spans. Variant scoring averages five independently trained models; these are not five different assay results.","status":"source_checked","source_ids":["evidence-official-9b820532ba8e3965f64e","evidence-official-ded281404bd1a0f3fdb7"],"source_locator":"Jaganathan et al., Cell 2019, STAR Methods: SpliceAI architecture and Model training and testing (pp. e2–e3); README.md: usage and License"},{"label":"Training data","value":"Human GRCh37 sequence and GENCODE V24lift37 principal protein-coding transcripts, split by chromosome with non-paralogous held-out test genes. The paper distinguishes GENCODE-only training from GTEx-junction-augmented models used for variant analyses.","status":"source_checked","source_ids":["evidence-official-9b820532ba8e3965f64e","evidence-official-ded281404bd1a0f3fdb7"],"source_locator":"Jaganathan et al., Cell 2019, STAR Methods: SpliceAI architecture and Model training and testing (pp. e2–e3); README.md: usage and License"},{"label":"Training cutoff","value":"GENCODE V24lift37 on GRCh37 defines the documented transcript annotations. GTEx-augmented training is separately described; the paper does not give one common latest-data date for both variants.","status":"source_checked","source_ids":["evidence-official-9b820532ba8e3965f64e","evidence-official-ded281404bd1a0f3fdb7"],"source_locator":"Jaganathan et al., Cell 2019, STAR Methods: SpliceAI architecture and Model training and testing (pp. e2–e3); README.md: usage and License"},{"label":"Context limits","value":"SpliceAI-10k uses 5,000 flanking bases on each side. An input of length l + 10,000 produces predictions for l central positions; receptive span is distinct from maximum input length.","status":"source_checked","source_ids":["evidence-official-9b820532ba8e3965f64e","evidence-official-ded281404bd1a0f3fdb7"],"source_locator":"Jaganathan et al., Cell 2019, STAR Methods: SpliceAI architecture and Model training and testing (pp. e2–e3); README.md: usage and License"},{"label":"Weights licence","value":"CC-BY-NC-4.0 for trained models; commercial use requires a separate licence. Code is PolyForm Strict 1.0.0.","status":"source_checked","source_ids":["evidence-official-9b820532ba8e3965f64e","evidence-official-ded281404bd1a0f3fdb7"],"source_locator":"Jaganathan et al., Cell 2019, STAR Methods: SpliceAI architecture and Model training and testing (pp. e2–e3); README.md: usage and License"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/illumina/SpliceAI","status":"source_checked","source_ids":["evidence-official-9b820532ba8e3965f64e","evidence-official-ded281404bd1a0f3fdb7"],"source_locator":"Jaganathan et al., Cell 2019, STAR Methods: SpliceAI architecture and Model training and testing (pp. e2–e3); README.md: usage and License"},{"label":"Code licence","value":"PolyForm Strict 1.0.0 for code; trained weights have separate terms.","status":"source_checked","source_ids":["evidence-official-9a47f0326b33870b3113"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"Provides direct sequence inference and an annotation workflow with explicit genome and distance settings.","source_ids":["evidence-official-9b820532ba8e3965f64e","evidence-official-ded281404bd1a0f3fdb7"],"source_locator":"Jaganathan et al., Cell 2019, STAR Methods: SpliceAI architecture and Model training and testing (pp. e2–e3); README.md: usage and License"}],"limitations":[{"text":"The command-line pipeline skips unsupported variants and variants outside its gene annotations. Code, model weights and downloadable precomputed scores have distinct licensing provisions.","source_ids":["evidence-official-9b820532ba8e3965f64e","evidence-official-ded281404bd1a0f3fdb7"],"source_locator":"Jaganathan et al., Cell 2019, STAR Methods: SpliceAI architecture and Model training and testing (pp. e2–e3); README.md: usage and License"}],"diagram":{"title":"SpliceAI workflow","steps":["Variant plus sequence context","Reference and alternate predictions","Splice-site differences","Gain/loss annotations"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-9b820532ba8e3965f64e","evidence-official-ded281404bd1a0f3fdb7"],"source_locator":"Jaganathan et al., Cell 2019, STAR Methods: SpliceAI architecture and Model training and testing (pp. e2–e3); README.md: usage and License"},"coverage":"limited","gaps":["Parameters: The original STAR Methods specifies residual blocks, dilation and receptive spans, but does not state a complete parameter count for each released five-model scoring ensemble."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}},"description":"Splicing effect prediction","facets":{"areas":["genomics"]},"id":"discovery-model-spliceai","kind":"model","links":[],"name":"SpliceAI","source_ids":["src-discovery-illumina-spliceai"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-virtual-cell-challenge-2026"],"entity_level":"family","reported_name":"STATE","version":null,"historical_missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"State separates cellular representation learning from prediction of responses to perturbation.","summary_source_ids":["evidence-official-1df5b1865861177c0c75"],"summary_source_locator":"README.md: Getting started, State Transition Model, ST TOML configuration and Licenses","sections":[{"title":"How it works","body":"State separates cellular representation learning from prediction of responses to perturbation. State Embedding and State Transition are separate components; the documented transition workflow trains on specified expression features and perturbation metadata. The documented inputs are annData expression measurements, gene/cell-type labels and an explicit dataset/split configuration. The output consists of cell embeddings or predicted perturbed-expression matrices, depending on the selected component.","source_ids":["evidence-official-1df5b1865861177c0c75"],"source_locator":"README.md: Getting started, State Transition Model, ST TOML configuration and Licenses"},{"title":"Versions and reproducibility","body":"State Embedding (SE) and State Transition (ST), with separately trained checkpoints. The applicable input limits require configuration-specific checking.","source_ids":["evidence-official-1df5b1865861177c0c75"],"source_locator":"README.md: Getting started, State Transition Model, ST TOML configuration and Licenses"}],"facts":[{"label":"Model type","value":"Cell representation and perturbation-transition model family","status":"source_checked","source_ids":["evidence-official-1df5b1865861177c0c75"],"source_locator":"README.md: Getting started, State Transition Model, ST TOML configuration and Licenses"},{"label":"Architecture","value":"State Embedding and State Transition are separate components; the documented transition workflow trains on specified expression features and perturbation metadata.","status":"source_checked","source_ids":["evidence-official-1df5b1865861177c0c75"],"source_locator":"README.md: Getting started, State Transition Model, ST TOML configuration and Licenses"},{"label":"Inputs","value":"AnnData expression measurements, gene/cell-type labels and an explicit dataset/split configuration.","status":"source_checked","source_ids":["evidence-official-1df5b1865861177c0c75"],"source_locator":"README.md: Getting started, State Transition Model, ST TOML configuration and Licenses"},{"label":"Outputs","value":"Cell embeddings or predicted perturbed-expression matrices, depending on the selected component.","status":"source_checked","source_ids":["evidence-official-1df5b1865861177c0c75"],"source_locator":"README.md: Getting started, State Transition Model, ST TOML configuration and Licenses"},{"label":"Parameters","value":"The repository contains separate State Embedding and State Transition configurations; the selected complete pipeline is required before a parameter total can be assigned.","status":"unreported","source_ids":["evidence-official-1df5b1865861177c0c75"],"source_locator":"README.md: Getting started, State Transition Model, ST TOML configuration and Licenses"},{"label":"Known versions","value":"State Embedding (SE) and State Transition (ST), with separately trained checkpoints.","status":"source_checked","source_ids":["evidence-official-1df5b1865861177c0c75"],"source_locator":"README.md: Getting started, State Transition Model, ST TOML configuration and Licenses"},{"label":"Training data","value":"Examples include Replogle–Nadig genetic perturbations and Tahoe-100M; the actual trained model depends on its configuration.","status":"source_checked","source_ids":["evidence-official-1df5b1865861177c0c75"],"source_locator":"README.md: Getting started, State Transition Model, ST TOML configuration and Licenses"},{"label":"Training cutoff","value":"The reviewed README and training configuration do not provide one latest-data date shared by all State Embedding and task-fitted State Transition releases.","status":"unreported","source_ids":["evidence-official-1df5b1865861177c0c75"],"source_locator":"README.md: Getting started, State Transition Model, ST TOML configuration and Licenses"},{"label":"Context limits","value":"The transition workflow operates on the chosen expression features and perturbation metadata; the inspected family documentation does not define one universal gene/cell token budget.","status":"unreported","source_ids":["evidence-official-1df5b1865861177c0c75"],"source_locator":"README.md: Getting started, State Transition Model, ST TOML configuration and Licenses"},{"label":"Weights licence","value":"Arc Research Institute State Model Non-Commercial License, with Acceptable Use Policy; separate from code CC-BY-NC-SA-4.0.","status":"source_checked","source_ids":["evidence-official-1df5b1865861177c0c75"],"source_locator":"README.md: Getting started, State Transition Model, ST TOML configuration and Licenses"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/ArcInstitute/state","status":"source_checked","source_ids":["evidence-official-1df5b1865861177c0c75"],"source_locator":"README.md: Getting started, State Transition Model, ST TOML configuration and Licenses"},{"label":"Code licence","value":"CC-BY-NC-SA-4.0","status":"source_checked","source_ids":["evidence-official-ed0382dc026ea7080abd"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"The command-line workflow makes training, inference and evaluation configurations explicit.","source_ids":["evidence-official-1df5b1865861177c0c75"],"source_locator":"README.md: Getting started, State Transition Model, ST TOML configuration and Licenses"}],"limitations":[{"text":"State Embedding and State Transition are not interchangeable model identities. Data omitted from zero-shot/few-shot split declarations default to training in the documented configuration format.","source_ids":["evidence-official-1df5b1865861177c0c75"],"source_locator":"README.md: Getting started, State Transition Model, ST TOML configuration and Licenses"}],"diagram":{"title":"STATE workflow","steps":["Expression and perturbation metadata","Selected embedding or transition component","Configured inference","Embeddings or predicted response"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-1df5b1865861177c0c75"],"source_locator":"README.md: Getting started, State Transition Model, ST TOML configuration and Licenses"},"coverage":"limited","gaps":["Parameters: The repository contains separate State Embedding and State Transition configurations; the selected complete pipeline is required before a parameter total can be assigned.","Training cutoff: The reviewed README and training configuration do not provide one latest-data date shared by all State Embedding and task-fitted State Transition releases.","Context limits: The transition workflow operates on the chosen expression features and perturbation metadata; the inspected family documentation does not define one universal gene/cell token budget."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}},"description":"Cell state and perturbation modelling","facets":{"areas":["single-cell"]},"id":"discovery-model-state","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-virtual-cell-challenge-2026"}],"name":"STATE","source_ids":["src-discovery-arcinstitute-state"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-glycanml"],"entity_level":"family","reported_name":"SweetNet","version":null,"historical_missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"SweetNet predicts glycan properties and produces learned representations from glycan graphs.","summary_source_ids":["evidence-official-7c43b700aa87c22843a6","evidence-official-2bf115f7d072744af00a"],"summary_source_locator":"glycowork/ml/models.py: SweetNet.__init__ and forward; README.md: pretrained SweetNet description","sections":[{"title":"How it works","body":"SweetNet predicts glycan properties and produces learned representations from glycan graphs. Graph convolutional network; the inspected implementation has three graph-convolution layers, global mean pooling and fully connected prediction layers. The documented inputs are tokenized glycan graph nodes and glycosidic connectivity. The output consists of property predictions and optional intermediate glycan representations.","source_ids":["evidence-official-7c43b700aa87c22843a6","evidence-official-2bf115f7d072744af00a"],"source_locator":"glycowork/ml/models.py: SweetNet.__init__ and forward; README.md: pretrained SweetNet description"},{"title":"Versions and reproducibility","body":"SweetNet class in the pinned glycowork revision; checkpoint/task identity remains separate. The applicable input limits require configuration-specific checking.","source_ids":["evidence-official-7c43b700aa87c22843a6","evidence-official-2bf115f7d072744af00a"],"source_locator":"glycowork/ml/models.py: SweetNet.__init__ and forward; README.md: pretrained SweetNet description"}],"facts":[{"label":"Model type","value":"Glycan graph convolutional network","status":"source_checked","source_ids":["evidence-official-7c43b700aa87c22843a6","evidence-official-2bf115f7d072744af00a"],"source_locator":"glycowork/ml/models.py: SweetNet.__init__ and forward; README.md: pretrained SweetNet description"},{"label":"Architecture","value":"Graph convolutional network; the inspected implementation has three graph-convolution layers, global mean pooling and fully connected prediction layers.","status":"source_checked","source_ids":["evidence-official-7c43b700aa87c22843a6","evidence-official-2bf115f7d072744af00a"],"source_locator":"glycowork/ml/models.py: SweetNet.__init__ and forward; README.md: pretrained SweetNet description"},{"label":"Inputs","value":"Tokenized glycan graph nodes and glycosidic connectivity.","status":"source_checked","source_ids":["evidence-official-7c43b700aa87c22843a6","evidence-official-2bf115f7d072744af00a"],"source_locator":"glycowork/ml/models.py: SweetNet.__init__ and forward; README.md: pretrained SweetNet description"},{"label":"Outputs","value":"Property predictions and optional intermediate glycan representations.","status":"source_checked","source_ids":["evidence-official-7c43b700aa87c22843a6","evidence-official-2bf115f7d072744af00a"],"source_locator":"glycowork/ml/models.py: SweetNet.__init__ and forward; README.md: pretrained SweetNet description"},{"label":"Parameters","value":"Depends on vocabulary size, hidden dimension and output classes; default hidden dimension in the inspected class is 128.","status":"source_checked","source_ids":["evidence-official-7c43b700aa87c22843a6","evidence-official-2bf115f7d072744af00a"],"source_locator":"glycowork/ml/models.py: SweetNet.__init__ and forward; README.md: pretrained SweetNet description"},{"label":"Known versions","value":"SweetNet class in the pinned glycowork revision; checkpoint/task identity remains separate.","status":"source_checked","source_ids":["evidence-official-7c43b700aa87c22843a6","evidence-official-2bf115f7d072744af00a"],"source_locator":"glycowork/ml/models.py: SweetNet.__init__ and forward; README.md: pretrained SweetNet description"},{"label":"Training data","value":"The inspected repository describes pretrained glycan-to-species prediction, but does not identify the exact training snapshot for that downloadable model.","status":"unreported","source_ids":["evidence-official-7c43b700aa87c22843a6","evidence-official-2bf115f7d072744af00a"],"source_locator":"glycowork/ml/models.py: SweetNet.__init__ and forward; README.md: pretrained SweetNet description"},{"label":"Training cutoff","value":"The reviewed pretrained-model documentation does not state the last-included glycan or species annotation date for the checkpoint.","status":"unreported","source_ids":["evidence-official-7c43b700aa87c22843a6","evidence-official-2bf115f7d072744af00a"],"source_locator":"glycowork/ml/models.py: SweetNet.__init__ and forward; README.md: pretrained SweetNet description"},{"label":"Context limits","value":"SweetNet pools variable-size glycan graphs. The inspected model implementation does not declare one validated maximum graph size for the pretrained checkpoint.","status":"unreported","source_ids":["evidence-official-7c43b700aa87c22843a6","evidence-official-2bf115f7d072744af00a"],"source_locator":"glycowork/ml/models.py: SweetNet.__init__ and forward; README.md: pretrained SweetNet description"},{"label":"Weights licence","value":"Separate checkpoint-distribution terms are not stated in the inspected release documentation and licence material. The source-code licence alone is not recorded as an explicit weight grant.","status":"unreported","source_ids":["evidence-official-7c43b700aa87c22843a6","evidence-official-2bf115f7d072744af00a","evidence-official-6bcab1b3e31b52c38ea5"],"source_locator":"glycowork/ml/models.py: SweetNet.__init__ and forward; README.md: pretrained SweetNet description; LICENSE: licence text"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/BojarLab/glycowork","status":"source_checked","source_ids":["evidence-official-7c43b700aa87c22843a6","evidence-official-2bf115f7d072744af00a"],"source_locator":"glycowork/ml/models.py: SweetNet.__init__ and forward; README.md: pretrained SweetNet description"},{"label":"Code licence","value":"MIT","status":"source_checked","source_ids":["evidence-official-6bcab1b3e31b52c38ea5"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"Models branched glycan connectivity directly instead of treating each glycan only as a linear string.","source_ids":["evidence-official-7c43b700aa87c22843a6","evidence-official-2bf115f7d072744af00a"],"source_locator":"glycowork/ml/models.py: SweetNet.__init__ and forward; README.md: pretrained SweetNet description"}],"limitations":[{"text":"The package contains several other models, including LectinOracle, whose protein inputs must not be assigned to SweetNet. The class configuration and pretrained task identify the actual model.","source_ids":["evidence-official-7c43b700aa87c22843a6","evidence-official-2bf115f7d072744af00a"],"source_locator":"glycowork/ml/models.py: SweetNet.__init__ and forward; README.md: pretrained SweetNet description"}],"diagram":{"title":"SweetNet workflow","steps":["Glycan graph","Three graph convolutions","Graph pooling","Property prediction"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-7c43b700aa87c22843a6","evidence-official-2bf115f7d072744af00a"],"source_locator":"glycowork/ml/models.py: SweetNet.__init__ and forward; README.md: pretrained SweetNet description"},"coverage":"limited","gaps":["Training data: The inspected repository describes pretrained glycan-to-species prediction, but does not identify the exact training snapshot for that downloadable model.","Training cutoff: The reviewed pretrained-model documentation does not state the last-included glycan or species annotation date for the checkpoint.","Context limits: SweetNet pools variable-size glycan graphs. The inspected model implementation does not declare one validated maximum graph size for the pretrained checkpoint.","Weights licence: Separate checkpoint-distribution terms are not stated in the inspected release documentation and licence material. The source-code licence alone is not recorded as an explicit weight grant."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}},"description":"Glycan graph learning model","facets":{"areas":["glycomics"]},"id":"discovery-model-sweetnet","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-glycanml"}],"name":"SweetNet","source_ids":["src-discovery-bojarlab-glycowork"],"status":"discovered"} {"attributes":{"entity_level":"method","reported_name":"Bepler","version":null,"historical_missing_metadata":{"checkpoint":"unreported","version":"unreported"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"The TAPE Bepler comparison uses a protein representation that combines bidirectional language modelling with supervised structural pretraining.","summary_source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-116dfd8f61e04b51c148","evidence-official-362b9cef071111dfe343"],"summary_source_locator":"README.md: opening reproducibility warning and Leaderboard; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3","sections":[{"title":"How it works","body":"The TAPE Bepler comparison uses a protein representation that combines bidirectional language modelling with supervised structural pretraining. The June 2019 TAPE preprint describes a two-layer bidirectional language model followed by three 512-unit bidirectional LSTMs, with contact and remote-homology supervision. The documented inputs are protein sequences for the task-specific TAPE evaluation. The output consists of protein representations followed by the applicable TAPE prediction head.","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-116dfd8f61e04b51c148","evidence-official-362b9cef071111dfe343"],"source_locator":"README.md: opening reproducibility warning and Leaderboard; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"title":"Versions and reproducibility","body":"June 2019 TAPE preprint configurations and the later PyTorch package are distinct; the current README warns it is not an exact reproduction. The TAPE preprint uses sequence-length-dependent batching; the exact task/checkpoint determines padding, truncation and resource constraints.","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-116dfd8f61e04b51c148","evidence-official-362b9cef071111dfe343"],"source_locator":"README.md: opening reproducibility warning and Leaderboard; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"}],"facts":[{"label":"Model type","value":"Bidirectional recurrent protein representation with structural pretraining","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-116dfd8f61e04b51c148","evidence-official-362b9cef071111dfe343"],"source_locator":"README.md: opening reproducibility warning and Leaderboard; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Architecture","value":"The June 2019 TAPE preprint describes a two-layer bidirectional language model followed by three 512-unit bidirectional LSTMs, with contact and remote-homology supervision.","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-116dfd8f61e04b51c148","evidence-official-362b9cef071111dfe343"],"source_locator":"README.md: opening reproducibility warning and Leaderboard; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Inputs","value":"Protein sequences for the task-specific TAPE evaluation.","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-116dfd8f61e04b51c148","evidence-official-362b9cef071111dfe343"],"source_locator":"README.md: opening reproducibility warning and Leaderboard; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Outputs","value":"Protein representations followed by the applicable TAPE prediction head.","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-116dfd8f61e04b51c148","evidence-official-362b9cef071111dfe343"],"source_locator":"README.md: opening reproducibility warning and Leaderboard; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Parameters","value":"The paper describes the language model and structural LSTMs but the exact historical checkpoint remains unresolved; no single verified total is assigned.","status":"unreported","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-116dfd8f61e04b51c148","evidence-official-362b9cef071111dfe343"],"source_locator":"README.md: opening reproducibility warning and Leaderboard; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Known versions","value":"June 2019 TAPE preprint configurations and the later PyTorch package are distinct; the current README warns it is not an exact reproduction.","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-116dfd8f61e04b51c148","evidence-official-362b9cef071111dfe343"],"source_locator":"README.md: opening reproducibility warning and Leaderboard; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Training data","value":"The June 2019 TAPE preprint uses 31M Pfam domains, with held-out families and a separate random split; downstream task heads are fitted on their own task data.","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-116dfd8f61e04b51c148","evidence-official-362b9cef071111dfe343"],"source_locator":"README.md: opening reproducibility warning and Leaderboard; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Training cutoff","value":"The June 2019 TAPE preprint identifies the Pfam-domain corpus and split procedure but does not state one latest-sequence deposition date. Later package defaults are a different implementation.","status":"unreported","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-116dfd8f61e04b51c148","evidence-official-362b9cef071111dfe343"],"source_locator":"README.md: opening reproducibility warning and Leaderboard; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Context limits","value":"The TAPE preprint uses sequence-length-dependent batching; the exact task/checkpoint determines padding, truncation and resource constraints.","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-116dfd8f61e04b51c148","evidence-official-362b9cef071111dfe343"],"source_locator":"README.md: opening reproducibility warning and Leaderboard; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Weights licence","value":"Separate checkpoint-distribution terms are not stated in the inspected release documentation and licence material. The source-code licence alone is not recorded as an explicit weight grant.","status":"unreported","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-116dfd8f61e04b51c148","evidence-official-362b9cef071111dfe343","evidence-official-e847949b10d83e3b0efe"],"source_locator":"README.md: opening reproducibility warning and Leaderboard; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3; LICENSE: licence text"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/songlab-cal/tape","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-116dfd8f61e04b51c148","evidence-official-362b9cef071111dfe343"],"source_locator":"README.md: opening reproducibility warning and Leaderboard; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Code licence","value":"BSD-3-Clause","status":"source_checked","source_ids":["evidence-official-e847949b10d83e3b0efe"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"Provides a reference for testing whether structural supervision during pretraining transfers to downstream protein tasks.","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-116dfd8f61e04b51c148","evidence-official-362b9cef071111dfe343"],"source_locator":"README.md: opening reproducibility warning and Leaderboard; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"}],"limitations":[{"text":"The preprint identifies the architecture and training objectives, but the exact historical checkpoint is still unresolved. Later PyTorch package defaults are not an exact reproduction.","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-116dfd8f61e04b51c148","evidence-official-362b9cef071111dfe343"],"source_locator":"README.md: opening reproducibility warning and Leaderboard; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"}],"diagram":{"title":"TAPE Bepler workflow","steps":["Protein sequence","Bidirectional language model","Three bidirectional LSTMs","Specified downstream task head"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-116dfd8f61e04b51c148","evidence-official-362b9cef071111dfe343"],"source_locator":"README.md: opening reproducibility warning and Leaderboard; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},"coverage":"limited","gaps":["Parameters: The paper describes the language model and structural LSTMs but the exact historical checkpoint remains unresolved; no single verified total is assigned.","Training cutoff: The June 2019 TAPE preprint identifies the Pfam-domain corpus and split procedure but does not state one latest-sequence deposition date. Later package defaults are a different implementation.","Weights licence: Separate checkpoint-distribution terms are not stated in the inspected release documentation and licence material. The source-code licence alone is not recorded as an explicit weight grant.","The historical Bepler checkpoint/variant is unresolved; no architecture equivalence or result aggregation is asserted."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}},"description":"Method identity as printed in the historical TAPE leaderboard.","facets":{"areas":["protein-function"]},"id":"discovery-model-tape-bepler","kind":"model","links":[],"name":"TAPE Bepler","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"entity_level":"method","reported_name":"LSTM","version":null,"historical_missing_metadata":{"checkpoint":"unreported","version":"unreported"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"The TAPE LSTM baseline represents protein sequences using recurrent networks that read residues in both directions.","summary_source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"summary_source_locator":"tape/models/modeling_lstm.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3","sections":[{"title":"How it works","body":"The TAPE LSTM baseline represents protein sequences using recurrent networks that read residues in both directions. Bidirectional recurrent protein encoder with three forward and three reverse LSTM layers in the default implementation. The documented inputs are amino-acid sequence in the tokenizer expected by the selected implementation. The output consists of residue/sequence representations and predictions from the chosen task head.","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_lstm.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"title":"Versions and reproducibility","body":"June 2019 TAPE preprint configurations and the later PyTorch package are distinct; the current README warns it is not an exact reproduction. The TAPE preprint uses sequence-length-dependent batching; the exact task/checkpoint determines padding, truncation and resource constraints.","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_lstm.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"}],"facts":[{"label":"Model type","value":"Bidirectional recurrent protein encoder","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_lstm.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Architecture","value":"Bidirectional recurrent protein encoder with three forward and three reverse LSTM layers in the default implementation.","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_lstm.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Inputs","value":"Amino-acid sequence in the tokenizer expected by the selected implementation.","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_lstm.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Outputs","value":"Residue/sequence representations and predictions from the chosen task head.","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_lstm.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Parameters","value":"Default input embedding size 128 and recurrent hidden size 1,024; these are dimensions, not parameter totals.","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_lstm.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Known versions","value":"June 2019 TAPE preprint configurations and the later PyTorch package are distinct; the current README warns it is not an exact reproduction.","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_lstm.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Training data","value":"The June 2019 TAPE preprint uses 31M Pfam domains, with held-out families and a separate random split; downstream task heads are fitted on their own task data.","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_lstm.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Training cutoff","value":"The June 2019 TAPE preprint identifies the Pfam-domain corpus and split procedure but does not state one latest-sequence deposition date. Later package defaults are a different implementation.","status":"unreported","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_lstm.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Context limits","value":"The TAPE preprint uses sequence-length-dependent batching; the exact task/checkpoint determines padding, truncation and resource constraints.","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_lstm.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Weights licence","value":"Separate checkpoint-distribution terms are not stated in the inspected release documentation and licence material. The source-code licence alone is not recorded as an explicit weight grant.","status":"unreported","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343","evidence-official-e847949b10d83e3b0efe"],"source_locator":"tape/models/modeling_lstm.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3; LICENSE: licence text"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/songlab-cal/tape","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_lstm.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Code licence","value":"BSD-3-Clause","status":"source_checked","source_ids":["evidence-official-e847949b10d83e3b0efe"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"Provides a defined reference architecture that can be paired with the same downstream tasks as other protein representations.","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_lstm.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"}],"limitations":[{"text":"The maintainers explicitly state that the current PyTorch repository is not an exact reproduction of the original TensorFlow paper code; training maintenance was discontinued. Default input embedding size 128 and recurrent hidden size 1,024; these are dimensions, not parameter totals.","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_lstm.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"}],"diagram":{"title":"TAPE LSTM workflow","steps":["Protein sequence","Forward and reverse LSTMs","Residue/sequence representation","Task head"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_lstm.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},"coverage":"limited","gaps":["Training cutoff: The June 2019 TAPE preprint identifies the Pfam-domain corpus and split procedure but does not state one latest-sequence deposition date. Later package defaults are a different implementation.","Weights licence: Separate checkpoint-distribution terms are not stated in the inspected release documentation and licence material. The source-code licence alone is not recorded as an explicit weight grant."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}},"description":"Method identity as printed in the historical TAPE leaderboard.","facets":{"areas":["protein-function"]},"id":"discovery-model-tape-lstm","kind":"model","links":[],"name":"TAPE LSTM","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"entity_level":"method","reported_name":"One Hot","version":null,"historical_missing_metadata":{"checkpoint":"unreported","version":"unreported"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"The TAPE one-hot baseline encodes each amino acid directly, allowing task performance to be assessed without a pretrained sequence representation.","summary_source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"summary_source_locator":"tape/models/modeling_onehot.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3","sections":[{"title":"How it works","body":"The TAPE one-hot baseline encodes each amino acid directly, allowing task performance to be assessed without a pretrained sequence representation. Direct one-hot residue representation followed by a separately fitted task head. The documented inputs are amino-acid sequence in the tokenizer expected by the selected implementation. The output consists of residue/sequence representations and predictions from the chosen task head.","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_onehot.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"title":"Versions and reproducibility","body":"June 2019 TAPE preprint configurations and the later PyTorch package are distinct; the current README warns it is not an exact reproduction. Variable-length residue input in the implementation; padding/masking and the downstream head determine the evaluated length handling.","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_onehot.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"}],"facts":[{"label":"Model type","value":"One-hot sequence representation and separate task head","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_onehot.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Architecture","value":"Direct one-hot residue representation followed by a separately fitted task head.","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_onehot.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Inputs","value":"Amino-acid sequence in the tokenizer expected by the selected implementation.","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_onehot.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Outputs","value":"Residue/sequence representations and predictions from the chosen task head.","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_onehot.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Parameters","value":"The encoding is parameter-free; the task-specific prediction head can still have learned parameters.","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_onehot.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Known versions","value":"June 2019 TAPE preprint configurations and the later PyTorch package are distinct; the current README warns it is not an exact reproduction.","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_onehot.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Training data","value":"The repository distributes a Pfam pretraining corpus and separate supervised task data. A specific checkpoint and adaptation must identify which training was actually used.","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_onehot.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Training cutoff","value":"Inapplicable to the fixed one-hot encoding; supervised task-head training dates belong to each evaluation.","status":"inapplicable","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_onehot.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Context limits","value":"Variable-length residue input in the implementation; padding/masking and the downstream head determine the evaluated length handling.","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_onehot.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Weights licence","value":"Inapplicable to the parameter-free residue encoding; a fitted downstream head requires its own checkpoint provenance and terms.","status":"inapplicable","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_onehot.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/songlab-cal/tape","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_onehot.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Code licence","value":"BSD-3-Clause","status":"source_checked","source_ids":["evidence-official-e847949b10d83e3b0efe"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"Provides a defined reference architecture that can be paired with the same downstream tasks as other protein representations.","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_onehot.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"}],"limitations":[{"text":"The maintainers explicitly state that the current PyTorch repository is not an exact reproduction of the original TensorFlow paper code; training maintenance was discontinued. The encoding is parameter-free; the task-specific prediction head can still have learned parameters.","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_onehot.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"}],"diagram":{"title":"TAPE One Hot workflow","steps":["Protein sequence","One-hot residue vectors","Specified task head","Task predictions"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_onehot.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}},"description":"Method identity as printed in the historical TAPE leaderboard.","facets":{"areas":["protein-function"]},"id":"discovery-model-tape-one-hot","kind":"model","links":[],"name":"TAPE One Hot","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"entity_level":"method","reported_name":"ResNet","version":null,"historical_missing_metadata":{"checkpoint":"unreported","version":"unreported"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"The TAPE ResNet baseline represents protein sequences using residual convolutional blocks before a task-specific prediction head.","summary_source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"summary_source_locator":"tape/models/modeling_resnet.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3","sections":[{"title":"How it works","body":"The TAPE ResNet baseline represents protein sequences using residual convolutional blocks before a task-specific prediction head. The June 2019 preprint uses 35 residual blocks with two dilated convolutions each, 256 filters and kernel width 9. The later PyTorch defaults use30 layers and width 512. The documented inputs are amino-acid sequence in the tokenizer expected by the selected implementation. The output consists of residue/sequence representations and predictions from the chosen task head.","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_resnet.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"title":"Versions and reproducibility","body":"June 2019 TAPE preprint configurations and the later PyTorch package are distinct; the current README warns it is not an exact reproduction. The TAPE preprint uses sequence-length-dependent batching; the exact task/checkpoint determines padding, truncation and resource constraints.","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_resnet.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"}],"facts":[{"label":"Model type","value":"Residual convolutional protein encoder","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_resnet.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Architecture","value":"The June 2019 preprint uses 35 residual blocks with two dilated convolutions each, 256 filters and kernel width 9. The later PyTorch defaults use30 layers and width 512.","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_resnet.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Inputs","value":"Amino-acid sequence in the tokenizer expected by the selected implementation.","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_resnet.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Outputs","value":"Residue/sequence representations and predictions from the chosen task head.","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_resnet.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Parameters","value":"The model also exposes separate downstream heads; count parameters for the exact selected head/configuration.","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_resnet.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Known versions","value":"June 2019 TAPE preprint configurations and the later PyTorch package are distinct; the current README warns it is not an exact reproduction.","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_resnet.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Training data","value":"The June 2019 TAPE preprint uses 31M Pfam domains, with held-out families and a separate random split; downstream task heads are fitted on their own task data.","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_resnet.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Training cutoff","value":"The June 2019 TAPE preprint identifies the Pfam-domain corpus and split procedure but does not state one latest-sequence deposition date. Later package defaults are a different implementation.","status":"unreported","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_resnet.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Context limits","value":"The TAPE preprint uses sequence-length-dependent batching; the exact task/checkpoint determines padding, truncation and resource constraints.","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_resnet.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Weights licence","value":"Separate checkpoint-distribution terms are not stated in the inspected release documentation and licence material. The source-code licence alone is not recorded as an explicit weight grant.","status":"unreported","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343","evidence-official-e847949b10d83e3b0efe"],"source_locator":"tape/models/modeling_resnet.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3; LICENSE: licence text"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/songlab-cal/tape","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_resnet.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Code licence","value":"BSD-3-Clause","status":"source_checked","source_ids":["evidence-official-e847949b10d83e3b0efe"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"Provides a defined reference architecture that can be paired with the same downstream tasks as other protein representations.","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_resnet.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"}],"limitations":[{"text":"The maintainers explicitly state that the current PyTorch repository is not an exact reproduction of the original TensorFlow paper code; training maintenance was discontinued. The model also exposes separate downstream heads; count parameters for the exact selected head/configuration.","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_resnet.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"}],"diagram":{"title":"TAPE ResNet workflow","steps":["Protein sequence","Residue embeddings","Residual convolutions","Task head"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_resnet.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},"coverage":"limited","gaps":["Training cutoff: The June 2019 TAPE preprint identifies the Pfam-domain corpus and split procedure but does not state one latest-sequence deposition date. Later package defaults are a different implementation.","Weights licence: Separate checkpoint-distribution terms are not stated in the inspected release documentation and licence material. The source-code licence alone is not recorded as an explicit weight grant."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}},"description":"Method identity as printed in the historical TAPE leaderboard.","facets":{"areas":["protein-function"]},"id":"discovery-model-tape-resnet","kind":"model","links":[],"name":"TAPE ResNet","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"entity_level":"method","reported_name":"Transformer","version":null,"historical_missing_metadata":{"checkpoint":"unreported","version":"unreported"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"The TAPE Transformer learns contextual protein representations using masked-residue pretraining and a task-specific prediction head.","summary_source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"summary_source_locator":"tape/models/modeling_bert.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3","sections":[{"title":"How it works","body":"The TAPE Transformer learns contextual protein representations using masked-residue pretraining and a task-specific prediction head. The June 2019 preprint uses 12 transformer layers, width 512 and eight heads (38M parameters). The later PyTorch BertConfig defaults to width 768 and 12 heads; these are different configurations. The documented inputs are amino-acid sequence in the tokenizer expected by the selected implementation. The output consists of residue/sequence representations and predictions from the chosen task head.","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_bert.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"title":"Versions and reproducibility","body":"June 2019 TAPE preprint configurations and the later PyTorch package are distinct; the current README warns it is not an exact reproduction. The inspected default BertConfig sets max_position_embeddings to 8,096; this is an implementation default, not evidence that a historical TAPE checkpoint was trained at that length.","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_bert.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"}],"facts":[{"label":"Model type","value":"BERT-style protein transformer encoder","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_bert.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Architecture","value":"The June 2019 preprint uses 12 transformer layers, width 512 and eight heads (38M parameters). The later PyTorch BertConfig defaults to width 768 and 12 heads; these are different configurations.","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_bert.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Inputs","value":"Amino-acid sequence in the tokenizer expected by the selected implementation.","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_bert.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Outputs","value":"Residue/sequence representations and predictions from the chosen task head.","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_bert.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Parameters","value":"38M for the June 2019 preprint Transformer; do not apply that total to the later PyTorch defaults or every task head.","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_bert.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Known versions","value":"June 2019 TAPE preprint configurations and the later PyTorch package are distinct; the current README warns it is not an exact reproduction.","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_bert.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Training data","value":"The June 2019 TAPE preprint uses 31M Pfam domains, with held-out families and a separate random split; downstream task heads are fitted on their own task data.","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_bert.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Training cutoff","value":"The June 2019 TAPE preprint identifies the Pfam-domain corpus and split procedure but does not state one latest-sequence deposition date. Later package defaults are a different implementation.","status":"unreported","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_bert.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Context limits","value":"The inspected default BertConfig sets max_position_embeddings to 8,096; this is an implementation default, not evidence that a historical TAPE checkpoint was trained at that length.","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_bert.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Weights licence","value":"Separate checkpoint-distribution terms are not stated in the inspected release documentation and licence material. The source-code licence alone is not recorded as an explicit weight grant.","status":"unreported","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343","evidence-official-e847949b10d83e3b0efe"],"source_locator":"tape/models/modeling_bert.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3; LICENSE: licence text"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/songlab-cal/tape","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_bert.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Code licence","value":"BSD-3-Clause","status":"source_checked","source_ids":["evidence-official-e847949b10d83e3b0efe"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"Provides a defined reference architecture that can be paired with the same downstream tasks as other protein representations.","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_bert.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"}],"limitations":[{"text":"The maintainers explicitly state that the current PyTorch repository is not an exact reproduction of the original TensorFlow paper code; training maintenance was discontinued. The default implementation reserves 8,096 position embeddings; this is an implementation setting, not proof of training or evaluation at that length.","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_bert.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"}],"diagram":{"title":"TAPE Transformer workflow","steps":["Protein sequence","Residue and position embeddings","Transformer encoder","Task head"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_bert.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},"coverage":"limited","gaps":["Training cutoff: The June 2019 TAPE preprint identifies the Pfam-domain corpus and split procedure but does not state one latest-sequence deposition date. Later package defaults are a different implementation.","Weights licence: Separate checkpoint-distribution terms are not stated in the inspected release documentation and licence material. The source-code licence alone is not recorded as an explicit weight grant."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}},"description":"Method identity as printed in the historical TAPE leaderboard.","facets":{"areas":["protein-function"]},"id":"discovery-model-tape-transformer","kind":"model","links":[],"name":"TAPE Transformer","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"entity_level":"method","reported_name":"Unirep","version":null,"historical_missing_metadata":{"checkpoint":"unreported","version":"unreported"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"The TAPE UniRep baseline uses a multiplicative recurrent network to represent protein sequences for downstream tasks.","summary_source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"summary_source_locator":"tape/models/modeling_unirep.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3","sections":[{"title":"How it works","body":"The TAPE UniRep baseline uses a multiplicative recurrent network to represent protein sequences for downstream tasks. Multiplicative LSTM protein encoder; the babbler-1900 configuration has recurrent hidden size 1,900. The documented inputs are amino-acid sequence in the tokenizer expected by the selected implementation. The output consists of residue/sequence representations and predictions from the chosen task head.","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_unirep.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"title":"Versions and reproducibility","body":"June 2019 TAPE preprint configurations and the later PyTorch package are distinct; the current README warns it is not an exact reproduction. The TAPE preprint uses sequence-length-dependent batching; the exact task/checkpoint determines padding, truncation and resource constraints.","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_unirep.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"}],"facts":[{"label":"Model type","value":"Multiplicative-LSTM protein encoder","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_unirep.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Architecture","value":"Multiplicative LSTM protein encoder; the babbler-1900 configuration has recurrent hidden size 1,900.","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_unirep.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Inputs","value":"Amino-acid sequence in the tokenizer expected by the selected implementation.","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_unirep.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Outputs","value":"Residue/sequence representations and predictions from the chosen task head.","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_unirep.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Parameters","value":"UniRep uses a different vocabulary from the other TAPE models; use its matching tokenizer.","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_unirep.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Known versions","value":"June 2019 TAPE preprint configurations and the later PyTorch package are distinct; the current README warns it is not an exact reproduction.","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_unirep.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Training data","value":"The June 2019 TAPE preprint uses 31M Pfam domains, with held-out families and a separate random split; downstream task heads are fitted on their own task data.","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_unirep.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Training cutoff","value":"The June 2019 TAPE preprint identifies the Pfam-domain corpus and split procedure but does not state one latest-sequence deposition date. Later package defaults are a different implementation.","status":"unreported","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_unirep.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Context limits","value":"The TAPE preprint uses sequence-length-dependent batching; the exact task/checkpoint determines padding, truncation and resource constraints.","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_unirep.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Weights licence","value":"Separate checkpoint-distribution terms are not stated in the inspected release documentation and licence material. The source-code licence alone is not recorded as an explicit weight grant.","status":"unreported","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343","evidence-official-e847949b10d83e3b0efe"],"source_locator":"tape/models/modeling_unirep.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3; LICENSE: licence text"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/songlab-cal/tape","status":"source_checked","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_unirep.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},{"label":"Code licence","value":"BSD-3-Clause","status":"source_checked","source_ids":["evidence-official-e847949b10d83e3b0efe"],"source_locator":"LICENSE: licence text"}],"strengths":[{"text":"Provides a defined reference architecture that can be paired with the same downstream tasks as other protein representations.","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_unirep.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"}],"limitations":[{"text":"The maintainers explicitly state that the current PyTorch repository is not an exact reproduction of the original TensorFlow paper code; training maintenance was discontinued. UniRep uses a different vocabulary from the other TAPE models; use its matching tokenizer.","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_unirep.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"}],"diagram":{"title":"TAPE Unirep workflow","steps":["Protein sequence","UniRep-specific tokens","Multiplicative LSTM","Representation or task head"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-028db5eeb224def23456","evidence-official-1aae5efd1b27ae9c761a","evidence-official-3cec54ed6ded9cb0a092","evidence-official-044ad0df4e4fe48a2125","evidence-official-12d43bf3076a93fc4ee7","evidence-official-c303652f3d79132a00c2","evidence-official-362b9cef071111dfe343"],"source_locator":"tape/models/modeling_unirep.py: configuration and model classes; README.md: opening maintenance warning, Examples and Data; June 2019 TAPE preprint Sections 4.1,5 and AppendixA.3"},"coverage":"limited","gaps":["Training cutoff: The June 2019 TAPE preprint identifies the Pfam-domain corpus and split procedure but does not state one latest-sequence deposition date. Later package defaults are a different implementation.","Weights licence: Separate checkpoint-distribution terms are not stated in the inspected release documentation and licence material. The source-code licence alone is not recorded as an explicit weight grant."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}},"description":"Method identity as printed in the historical TAPE leaderboard.","facets":{"areas":["protein-function"]},"id":"discovery-model-tape-unirep","kind":"model","links":[],"name":"TAPE Unirep","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-beacon"],"entity_level":"method","reported_name":"ViennaRNA RNAfold","version":null,"historical_missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"metadata_review_scope":"historical_missing_metadata preserves the original discovery state. Current descriptive evidence and missingness are recorded in profile.facts; numerical-result review is separate.","profile":{"summary":"RNAfold predicts RNA secondary structure and thermodynamic ensemble quantities within the ViennaRNA package.","summary_source_ids":["evidence-official-1b2aa7207e3a05f297ab"],"summary_source_locator":"README.md: package capabilities, executable programs, Energy Parameters and License","sections":[{"title":"How it works","body":"RNAfold predicts RNA secondary structure and thermodynamic ensemble quantities within the ViennaRNA package. Energy-based RNA secondary-structure calculation; minimum-free-energy and partition-function modes are distinct outputs. The documented inputs are RNA nucleotide sequence with selected thermodynamic settings and supported constraints. The output consists of predicted secondary structure and, in ensemble mode, partition-function-derived probabilities.","source_ids":["evidence-official-1b2aa7207e3a05f297ab"],"source_locator":"README.md: package capabilities, executable programs, Energy Parameters and License"},{"title":"Versions and reproducibility","body":"RNAfold within ViennaRNA; software and energy-parameter set must be pinned. Implementation and memory dependent; the package README gives an upper representational length with an explicit large-memory caveat.","source_ids":["evidence-official-1b2aa7207e3a05f297ab"],"source_locator":"README.md: package capabilities, executable programs, Energy Parameters and License"}],"facts":[{"label":"Model type","value":"Thermodynamic RNA secondary-structure procedure","status":"source_checked","source_ids":["evidence-official-1b2aa7207e3a05f297ab"],"source_locator":"README.md: package capabilities, executable programs, Energy Parameters and License"},{"label":"Architecture","value":"Energy-based RNA secondary-structure calculation; minimum-free-energy and partition-function modes are distinct outputs.","status":"source_checked","source_ids":["evidence-official-1b2aa7207e3a05f297ab"],"source_locator":"README.md: package capabilities, executable programs, Energy Parameters and License"},{"label":"Inputs","value":"RNA nucleotide sequence with selected thermodynamic settings and supported constraints.","status":"source_checked","source_ids":["evidence-official-1b2aa7207e3a05f297ab"],"source_locator":"README.md: package capabilities, executable programs, Energy Parameters and License"},{"label":"Outputs","value":"Predicted secondary structure and, in ensemble mode, partition-function-derived probabilities.","status":"source_checked","source_ids":["evidence-official-1b2aa7207e3a05f297ab"],"source_locator":"README.md: package capabilities, executable programs, Energy Parameters and License"},{"label":"Parameters","value":"Thermodynamic parameter set and algorithm settings; not neural model parameters.","status":"source_checked","source_ids":["evidence-official-1b2aa7207e3a05f297ab"],"source_locator":"README.md: package capabilities, executable programs, Energy Parameters and License"},{"label":"Known versions","value":"RNAfold within ViennaRNA; software and energy-parameter set must be pinned.","status":"source_checked","source_ids":["evidence-official-1b2aa7207e3a05f297ab"],"source_locator":"README.md: package capabilities, executable programs, Energy Parameters and License"},{"label":"Training data","value":"Energy parameters are empirical model inputs, not a pretrained neural checkpoint.","status":"source_checked","source_ids":["evidence-official-1b2aa7207e3a05f297ab"],"source_locator":"README.md: package capabilities, executable programs, Energy Parameters and License"},{"label":"Training cutoff","value":"Inapplicable to neural pretraining; reference-database and input-data dates must be recorded for each run.","status":"inapplicable","source_ids":["evidence-official-1b2aa7207e3a05f297ab"],"source_locator":"README.md: package capabilities, executable programs, Energy Parameters and License"},{"label":"Context limits","value":"Implementation and memory dependent; the package README gives an upper representational length with an explicit large-memory caveat.","status":"source_checked","source_ids":["evidence-official-1b2aa7207e3a05f297ab"],"source_locator":"README.md: package capabilities, executable programs, Energy Parameters and License"},{"label":"Weights licence","value":"Inapplicable to neural weights; preserve the thermodynamic parameter-file version and licence.","status":"inapplicable","source_ids":["evidence-official-1b2aa7207e3a05f297ab"],"source_locator":"README.md: package capabilities, executable programs, Energy Parameters and License"},{"label":"Access","value":"Official project documentation and implementation: https://github.com/ViennaRNA/ViennaRNA","status":"source_checked","source_ids":["evidence-official-1b2aa7207e3a05f297ab"],"source_locator":"README.md: package capabilities, executable programs, Energy Parameters and License"},{"label":"Code licence","value":"ViennaRNA custom permissive licence; inspect the pinned COPYING terms.","status":"source_checked","source_ids":["evidence-official-4269946a73db94a15289"],"source_locator":"COPYING: licence text"}],"strengths":[{"text":"Provides an interpretable thermodynamic reference for comparison with learned RNA structure methods.","source_ids":["evidence-official-1b2aa7207e3a05f297ab"],"source_locator":"README.md: package capabilities, executable programs, Energy Parameters and License"}],"limitations":[{"text":"The package includes multiple programs with different tasks. RNAfold secondary structure must not be presented as three-dimensional RNA folding. Sequence length is constrained by memory and algorithmic cost.","source_ids":["evidence-official-1b2aa7207e3a05f297ab"],"source_locator":"README.md: package capabilities, executable programs, Energy Parameters and License"}],"diagram":{"title":"ViennaRNA RNAfold workflow","steps":["RNA sequence and energy settings","Secondary-structure energy calculation","Minimum-energy or ensemble mode","Structure and probabilities"],"caption":"Conceptual summary of the documented data flow; optional inputs and configured downstream stages must be reported for a reproducible evaluation.","source_ids":["evidence-official-1b2aa7207e3a05f297ab"],"source_locator":"README.md: package capabilities, executable programs, Energy Parameters and License"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected pinned official documentation, relevant implementation files and named primary-paper sections. Claims are limited to those artifacts. Remaining field extraction and identity conflicts are explicit; no new performance claims, model runs or human review are implied."}}},"description":"Thermodynamic RNA secondary structure prediction","facets":{"areas":["rna"]},"id":"discovery-model-viennarna-rnafold","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-beacon"}],"name":"ViennaRNA RNAfold","source_ids":["src-discovery-viennarna-viennarna"],"status":"discovered"} {"attributes":{"metric":"Spearman's rho","metric_direction":"higher","missing_metadata":{"denominator":"unextracted","seeds":"unreported","uncertainty":"unreported"},"numeric_value":"0.33","printed_value":"0.33","review":{"method":"automated table parse plus separate source read and manually enumerated row/value cross-check","notes":"All six rows checked for this task. Source checked, not experiment reproduced. Historical README does not resolve exact checkpoint or data release.","reviewed_at":"2026-09-16T10:35:25.806835+00:00","reviewer":"Codex research agent; no human review claimed"},"source_locator":"README.md > Leaderboard > Fluorescence; row Bepler; column Spearman's rho","uncertainty":null,"unit":"correlation"},"description":"Reported score transcribed from the complete historical task table.","facets":{"areas":["protein-function"]},"id":"discovery-result-tape-fluorescence-bepler-spearman-rho","kind":"result","links":[{"relation":"evaluation","target_id":"discovery-evaluation-tape-fluorescence-bepler-leaderboard-evaluation"}],"name":"TAPE Fluorescence Bepler Spearman rho","source_ids":["src-discovery-songlab-cal-tape"],"status":"source_checked"} {"attributes":{"metric":"Spearman's rho","metric_direction":"higher","missing_metadata":{"denominator":"unextracted","seeds":"unreported","uncertainty":"unreported"},"numeric_value":"0.67","printed_value":"0.67","review":{"method":"automated table parse plus separate source read and manually enumerated row/value cross-check","notes":"All six rows checked for this task. Source checked, not experiment reproduced. Historical README does not resolve exact checkpoint or data release.","reviewed_at":"2026-09-16T10:35:25.806835+00:00","reviewer":"Codex research agent; no human review claimed"},"source_locator":"README.md > Leaderboard > Fluorescence; row LSTM; column Spearman's rho","uncertainty":null,"unit":"correlation"},"description":"Reported score transcribed from the complete historical task table.","facets":{"areas":["protein-function"]},"id":"discovery-result-tape-fluorescence-lstm-spearman-rho","kind":"result","links":[{"relation":"evaluation","target_id":"discovery-evaluation-tape-fluorescence-lstm-leaderboard-evaluation"}],"name":"TAPE Fluorescence LSTM Spearman rho","source_ids":["src-discovery-songlab-cal-tape"],"status":"source_checked"} {"attributes":{"metric":"Spearman's rho","metric_direction":"higher","missing_metadata":{"denominator":"unextracted","seeds":"unreported","uncertainty":"unreported"},"numeric_value":"0.14","printed_value":"0.14","review":{"method":"automated table parse plus separate source read and manually enumerated row/value cross-check","notes":"All six rows checked for this task. Source checked, not experiment reproduced. Historical README does not resolve exact checkpoint or data release.","reviewed_at":"2026-09-16T10:35:25.806835+00:00","reviewer":"Codex research agent; no human review claimed"},"source_locator":"README.md > Leaderboard > Fluorescence; row One Hot; column Spearman's rho","uncertainty":null,"unit":"correlation"},"description":"Reported score transcribed from the complete historical task table.","facets":{"areas":["protein-function"]},"id":"discovery-result-tape-fluorescence-one-hot-spearman-rho","kind":"result","links":[{"relation":"evaluation","target_id":"discovery-evaluation-tape-fluorescence-one-hot-leaderboard-evaluation"}],"name":"TAPE Fluorescence One Hot Spearman rho","source_ids":["src-discovery-songlab-cal-tape"],"status":"source_checked"} {"attributes":{"metric":"Spearman's rho","metric_direction":"higher","missing_metadata":{"denominator":"unextracted","seeds":"unreported","uncertainty":"unreported"},"numeric_value":"0.21","printed_value":"0.21","review":{"method":"automated table parse plus separate source read and manually enumerated row/value cross-check","notes":"All six rows checked for this task. Source checked, not experiment reproduced. Historical README does not resolve exact checkpoint or data release.","reviewed_at":"2026-09-16T10:35:25.806835+00:00","reviewer":"Codex research agent; no human review claimed"},"source_locator":"README.md > Leaderboard > Fluorescence; row ResNet; column Spearman's rho","uncertainty":null,"unit":"correlation"},"description":"Reported score transcribed from the complete historical task table.","facets":{"areas":["protein-function"]},"id":"discovery-result-tape-fluorescence-resnet-spearman-rho","kind":"result","links":[{"relation":"evaluation","target_id":"discovery-evaluation-tape-fluorescence-resnet-leaderboard-evaluation"}],"name":"TAPE Fluorescence ResNet Spearman rho","source_ids":["src-discovery-songlab-cal-tape"],"status":"source_checked"} {"attributes":{"metric":"Spearman's rho","metric_direction":"higher","missing_metadata":{"denominator":"unextracted","seeds":"unreported","uncertainty":"unreported"},"numeric_value":"0.68","printed_value":"0.68","review":{"method":"automated table parse plus separate source read and manually enumerated row/value cross-check","notes":"All six rows checked for this task. Source checked, not experiment reproduced. Historical README does not resolve exact checkpoint or data release.","reviewed_at":"2026-09-16T10:35:25.806835+00:00","reviewer":"Codex research agent; no human review claimed"},"source_locator":"README.md > Leaderboard > Fluorescence; row Transformer; column Spearman's rho","uncertainty":null,"unit":"correlation"},"description":"Reported score transcribed from the complete historical task table.","facets":{"areas":["protein-function"]},"id":"discovery-result-tape-fluorescence-transformer-spearman-rho","kind":"result","links":[{"relation":"evaluation","target_id":"discovery-evaluation-tape-fluorescence-transformer-leaderboard-evaluation"}],"name":"TAPE Fluorescence Transformer Spearman rho","source_ids":["src-discovery-songlab-cal-tape"],"status":"source_checked"} {"attributes":{"metric":"Spearman's rho","metric_direction":"higher","missing_metadata":{"denominator":"unextracted","seeds":"unreported","uncertainty":"unreported"},"numeric_value":"0.67","printed_value":"0.67","review":{"method":"automated table parse plus separate source read and manually enumerated row/value cross-check","notes":"All six rows checked for this task. Source checked, not experiment reproduced. Historical README does not resolve exact checkpoint or data release.","reviewed_at":"2026-09-16T10:35:25.806835+00:00","reviewer":"Codex research agent; no human review claimed"},"source_locator":"README.md > Leaderboard > Fluorescence; row Unirep; column Spearman's rho","uncertainty":null,"unit":"correlation"},"description":"Reported score transcribed from the complete historical task table.","facets":{"areas":["protein-function"]},"id":"discovery-result-tape-fluorescence-unirep-spearman-rho","kind":"result","links":[{"relation":"evaluation","target_id":"discovery-evaluation-tape-fluorescence-unirep-leaderboard-evaluation"}],"name":"TAPE Fluorescence Unirep Spearman rho","source_ids":["src-discovery-songlab-cal-tape"],"status":"source_checked"} {"attributes":{"metric":"Spearman's rho","metric_direction":"higher","missing_metadata":{"denominator":"unextracted","seeds":"unreported","uncertainty":"unreported"},"numeric_value":"0.64","printed_value":"0.64","review":{"method":"automated table parse plus separate source read and manually enumerated row/value cross-check","notes":"All six rows checked for this task. Source checked, not experiment reproduced. Historical README does not resolve exact checkpoint or data release.","reviewed_at":"2026-09-16T10:35:25.806835+00:00","reviewer":"Codex research agent; no human review claimed"},"source_locator":"README.md > Leaderboard > Stability; row Bepler; column Spearman's rho","uncertainty":null,"unit":"correlation"},"description":"Reported score transcribed from the complete historical task table.","facets":{"areas":["protein-function"]},"id":"discovery-result-tape-stability-bepler-spearman-rho","kind":"result","links":[{"relation":"evaluation","target_id":"discovery-evaluation-tape-stability-bepler-leaderboard-evaluation"}],"name":"TAPE Stability Bepler Spearman rho","source_ids":["src-discovery-songlab-cal-tape"],"status":"source_checked"} {"attributes":{"metric":"Spearman's rho","metric_direction":"higher","missing_metadata":{"denominator":"unextracted","seeds":"unreported","uncertainty":"unreported"},"numeric_value":"0.69","printed_value":"0.69","review":{"method":"automated table parse plus separate source read and manually enumerated row/value cross-check","notes":"All six rows checked for this task. Source checked, not experiment reproduced. Historical README does not resolve exact checkpoint or data release.","reviewed_at":"2026-09-16T10:35:25.806835+00:00","reviewer":"Codex research agent; no human review claimed"},"source_locator":"README.md > Leaderboard > Stability; row LSTM; column Spearman's rho","uncertainty":null,"unit":"correlation"},"description":"Reported score transcribed from the complete historical task table.","facets":{"areas":["protein-function"]},"id":"discovery-result-tape-stability-lstm-spearman-rho","kind":"result","links":[{"relation":"evaluation","target_id":"discovery-evaluation-tape-stability-lstm-leaderboard-evaluation"}],"name":"TAPE Stability LSTM Spearman rho","source_ids":["src-discovery-songlab-cal-tape"],"status":"source_checked"} {"attributes":{"metric":"Spearman's rho","metric_direction":"higher","missing_metadata":{"denominator":"unextracted","seeds":"unreported","uncertainty":"unreported"},"numeric_value":"0.19","printed_value":"0.19","review":{"method":"automated table parse plus separate source read and manually enumerated row/value cross-check","notes":"All six rows checked for this task. Source checked, not experiment reproduced. Historical README does not resolve exact checkpoint or data release.","reviewed_at":"2026-09-16T10:35:25.806835+00:00","reviewer":"Codex research agent; no human review claimed"},"source_locator":"README.md > Leaderboard > Stability; row One Hot; column Spearman's rho","uncertainty":null,"unit":"correlation"},"description":"Reported score transcribed from the complete historical task table.","facets":{"areas":["protein-function"]},"id":"discovery-result-tape-stability-one-hot-spearman-rho","kind":"result","links":[{"relation":"evaluation","target_id":"discovery-evaluation-tape-stability-one-hot-leaderboard-evaluation"}],"name":"TAPE Stability One Hot Spearman rho","source_ids":["src-discovery-songlab-cal-tape"],"status":"source_checked"} {"attributes":{"metric":"Spearman's rho","metric_direction":"higher","missing_metadata":{"denominator":"unextracted","seeds":"unreported","uncertainty":"unreported"},"numeric_value":"0.73","printed_value":"0.73","review":{"method":"automated table parse plus separate source read and manually enumerated row/value cross-check","notes":"All six rows checked for this task. Source checked, not experiment reproduced. Historical README does not resolve exact checkpoint or data release.","reviewed_at":"2026-09-16T10:35:25.806835+00:00","reviewer":"Codex research agent; no human review claimed"},"source_locator":"README.md > Leaderboard > Stability; row ResNet; column Spearman's rho","uncertainty":null,"unit":"correlation"},"description":"Reported score transcribed from the complete historical task table.","facets":{"areas":["protein-function"]},"id":"discovery-result-tape-stability-resnet-spearman-rho","kind":"result","links":[{"relation":"evaluation","target_id":"discovery-evaluation-tape-stability-resnet-leaderboard-evaluation"}],"name":"TAPE Stability ResNet Spearman rho","source_ids":["src-discovery-songlab-cal-tape"],"status":"source_checked"} {"attributes":{"metric":"Spearman's rho","metric_direction":"higher","missing_metadata":{"denominator":"unextracted","seeds":"unreported","uncertainty":"unreported"},"numeric_value":"0.73","printed_value":"0.73","review":{"method":"automated table parse plus separate source read and manually enumerated row/value cross-check","notes":"All six rows checked for this task. Source checked, not experiment reproduced. Historical README does not resolve exact checkpoint or data release.","reviewed_at":"2026-09-16T10:35:25.806835+00:00","reviewer":"Codex research agent; no human review claimed"},"source_locator":"README.md > Leaderboard > Stability; row Transformer; column Spearman's rho","uncertainty":null,"unit":"correlation"},"description":"Reported score transcribed from the complete historical task table.","facets":{"areas":["protein-function"]},"id":"discovery-result-tape-stability-transformer-spearman-rho","kind":"result","links":[{"relation":"evaluation","target_id":"discovery-evaluation-tape-stability-transformer-leaderboard-evaluation"}],"name":"TAPE Stability Transformer Spearman rho","source_ids":["src-discovery-songlab-cal-tape"],"status":"source_checked"} {"attributes":{"metric":"Spearman's rho","metric_direction":"higher","missing_metadata":{"denominator":"unextracted","seeds":"unreported","uncertainty":"unreported"},"numeric_value":"0.73","printed_value":"0.73","review":{"method":"automated table parse plus separate source read and manually enumerated row/value cross-check","notes":"All six rows checked for this task. Source checked, not experiment reproduced. Historical README does not resolve exact checkpoint or data release.","reviewed_at":"2026-09-16T10:35:25.806835+00:00","reviewer":"Codex research agent; no human review claimed"},"source_locator":"README.md > Leaderboard > Stability; row Unirep; column Spearman's rho","uncertainty":null,"unit":"correlation"},"description":"Reported score transcribed from the complete historical task table.","facets":{"areas":["protein-function"]},"id":"discovery-result-tape-stability-unirep-spearman-rho","kind":"result","links":[{"relation":"evaluation","target_id":"discovery-evaluation-tape-stability-unirep-leaderboard-evaluation"}],"name":"TAPE Stability Unirep Spearman rho","source_ids":["src-discovery-songlab-cal-tape"],"status":"source_checked"} {"id":"dna-foundation-models-2025","kind":"source","name":"Benchmarking DNA foundation models for genomic and genetic tasks","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["dna-genomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12663285/","version":"PMC12663285.1","retrieved_at":"2026-09-15T23:25:00Z","doi":"10.1038/s41467-025-65823-8","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"5d8ca9bcf88cc1b38ad667906a2e4699b1aefa6d31c6f49259784930353f3202","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12663285/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:35.327Z","legacy_paper":{"id":"dna-foundation-models-2025","title":"Benchmarking DNA foundation models for genomic and genetic tasks","year":2025,"publication_status":"peer_reviewed","version":"PMC12663285.1","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12663285/","primary_domain":"dna-genomes","retrieved_utc":"2026-09-15T23:25:00Z","notes":"Primary full text via Europe PMC XML; venue: Nature Communications; PMC ID: PMC12663285.","doi":"10.1038/s41467-025-65823-8"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"dnabert2-enhancer-2025","kind":"source","name":"Utilizing a deep learning model based on BERT for identifying enhancers and their strength","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["dna-genomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11981215/","version":"journal full text in PMC","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.1371/journal.pone.0320085","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"d052b80efe7bfc1380994ad28503a5575f04ef940f74d5c9c137cb4ba6827863","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11981215/fullTextXML","artifact_retrieved_at":"2026-09-16T10:38:57.558204+00:00","legacy_paper":{"id":"dnabert2-enhancer-2025","title":"Utilizing a deep learning model based on BERT for identifying enhancers and their strength","year":2025,"publication_status":"peer_reviewed","version":"journal full text in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11981215/","primary_domain":"dna-genomes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.1371/journal.pone.0320085","notes":"Numeric result checked against Table 4 in primary full-text XML; journal/source: PLOS One."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"dnalongbench-2025","kind":"source","name":"DNALongBench: A Benchmark Suite for Long-Range DNA Prediction Tasks","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["dna-genomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11741265/","version":"PMC11741265.1","retrieved_at":"2026-09-15T23:25:00Z","doi":"10.1101/2025.01.06.631595","publication_status":"preprint","year":2025,"artifact_sha256":"fa440a17cecf16a5d872d50a30910f7591b5f6f78e10a944c6bda5ea8d7e32dd","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11741265/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:16.492545+00:00","legacy_paper":{"id":"dnalongbench-2025","title":"DNALongBench: A Benchmark Suite for Long-Range DNA Prediction Tasks","year":2025,"publication_status":"preprint","version":"PMC11741265.1","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11741265/","primary_domain":"dna-genomes","retrieved_utc":"2026-09-15T23:25:00Z","notes":"Primary full text via Europe PMC XML; venue: bioRxiv; PMC ID: PMC11741265.","doi":"10.1101/2025.01.06.631595"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"eden-genomic-classification-2026","kind":"source","name":"EDEN: multiscale expected density of nucleotide encoding for enhanced DNA sequence classification with hybrid deep learning","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["dna-genomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12879454/","version":"journal full text in PMC","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.1186/s12859-026-06367-6","publication_status":"peer_reviewed","year":2026,"artifact_sha256":"38a6e26b3caffe8e021a2b0b672218e783aca9ee42046765e323946813015e65","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12879454/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:37.531Z","legacy_paper":{"id":"eden-genomic-classification-2026","title":"EDEN: multiscale expected density of nucleotide encoding for enhanced DNA sequence classification with hybrid deep learning","year":2026,"publication_status":"peer_reviewed","version":"journal full text in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12879454/","primary_domain":"dna-genomes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.1186/s12859-026-06367-6","notes":"DNABERT-2 comparator 70.52 is printed in Table 5. The article does not clearly document whether this comparator was independently rerun or consolidated from prior GUE results, so evaluation origin is conservatively marked paper_compilation."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"enbed-2024","kind":"source","name":"Understanding the natural language of DNA using encoder–decoder foundation models with byte-level precision","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["dna-genomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11341122/","version":"version of record","retrieved_at":"2026-09-15T23:25:00Z","doi":"10.1093/bioadv/vbae117","publication_status":"peer_reviewed","year":2024,"artifact_sha256":"e95d4be70d32e61af5a92eda8ea66f25a2cc629e3f83e7b5241b13cde8bdb83b","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11341122/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:35.378Z","legacy_paper":{"id":"enbed-2024","title":"Understanding the natural language of DNA using encoder–decoder foundation models with byte-level precision","year":2024,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11341122/","primary_domain":"dna-genomes","retrieved_utc":"2026-09-15T23:25:00Z","notes":"Primary full text via Europe PMC XML; venue: Bioinformatics Advances; PMC ID: PMC11341122.","doi":"10.1093/bioadv/vbae117"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"enhancer-position-encoding-2024","kind":"source","name":"A deep learning model for DNA enhancer prediction based on nucleotide position aware feature encoding","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["dna-genomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11167433/","version":"version of record","retrieved_at":"2026-09-15T23:37:05Z","doi":"10.1016/j.isci.2024.110030","publication_status":"peer_reviewed","year":2024,"artifact_sha256":"0183b6a111b1b02344cad35a571a1fd2c56257e406c5be1df69f7902c5d06749","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11167433/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:06Z","legacy_paper":{"id":"enhancer-position-encoding-2024","title":"A deep learning model for DNA enhancer prediction based on nucleotide position aware feature encoding","year":2024,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11167433/","primary_domain":"dna-genomes","retrieved_utc":"2026-09-15T23:37:05Z","notes":"Primary full text verified via Europe PMC fullTextXML; venue: iScience; PMC ID: PMC11167433. Task-specific CNN baseline, included as a DNA benchmark protocol reference.","doi":"10.1016/j.isci.2024.110030"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"ensemble-idp-docking-2025","kind":"source","name":"Ensemble docking for intrinsically disordered proteins","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11785235/","version":"preprint archived 2025-01-26","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.1101/2025.01.23.634614","publication_status":"preprint","year":2025,"artifact_sha256":"d02d91cdde41cb76ec5c86b532dffc564879c69e764a8c6b7752460fbbfd24b7","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11785235/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:56.275Z","legacy_paper":{"id":"ensemble-idp-docking-2025","title":"Ensemble docking for intrinsically disordered proteins","year":2025,"publication_status":"preprint","version":"preprint archived 2025-01-26","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11785235/","primary_domain":"molecular-interactions","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: bioRxiv; PMC ID: PMC11785235.","doi":"10.1101/2025.01.23.634614"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"ernie-rna-2025","kind":"source","name":"ERNIE-RNA: an RNA language model with structure-enhanced representations","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["rna-transcriptomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12627772/","version":"journal full text in PMC","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.1038/s41467-025-64972-0","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"0bd1d4b3cbf5d59d452cec4864614947861efcee050ba07e7de395cd90630047","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12627772/fullTextXML","artifact_retrieved_at":"2026-09-16T10:38:57.558206+00:00","legacy_paper":{"id":"ernie-rna-2025","title":"ERNIE-RNA: an RNA language model with structure-enhanced representations","year":2025,"publication_status":"peer_reviewed","version":"journal full text in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12627772/","primary_domain":"rna-transcriptomes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.1038/s41467-025-64972-0","notes":"Numeric result checked against Table 2 in primary full-text XML; journal/source: Nature Communications."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"esm2-amp-2025","kind":"source","name":"ESM2_AMP: an interpretable framework for protein–protein interactions prediction and biological mechanism discovery","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12392411/","version":"version of record","retrieved_at":"2026-09-15T23:37:05Z","doi":"10.1093/bib/bbaf434","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"8e7ad6efb72ca28d73037cdf465b0e62f99cd6d0ee4ca9eaf96a4c48da22fd6c","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12392411/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:58.625Z","legacy_paper":{"id":"esm2-amp-2025","title":"ESM2_AMP: an interpretable framework for protein–protein interactions prediction and biological mechanism discovery","year":2025,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12392411/","primary_domain":"molecular-interactions","retrieved_utc":"2026-09-15T23:37:05Z","notes":"Primary full text verified via Europe PMC fullTextXML; venue: Briefings in Bioinformatics; PMC ID: PMC12392411. Paper has multiple model variants; selected named ESM2_AMPS variant only.","doi":"10.1093/bib/bbaf434"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"esm2-ofs-fitness-2025","kind":"source","name":"Pseudo-perplexity in One Fell Swoop for Protein Fitness Estimation","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["proteins-complexes"]},"source_ids":[],"links":[],"attributes":{"url":"https://journals.aps.org/prxlife/pdf/10.1103/zhx7-hcmm","version":"PRX Life 2025 journal article","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.1103/zhx7-hcmm","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"085ef646f11b8e5335c4b3d86b15fb6c7bf5edf4a80a8b622753ac69d9991a67","artifact_url":"https://harvest.aps.org/v2/journals/articles/10.1103/zhx7-hcmm/fulltext","artifact_retrieved_at":"2026-09-16T10:45:41.099916+00:00","legacy_paper":{"id":"esm2-ofs-fitness-2025","title":"Pseudo-perplexity in One Fell Swoop for Protein Fitness Estimation","year":2025,"publication_status":"peer_reviewed","version":"PRX Life 2025 journal article","source_url":"https://journals.aps.org/prxlife/pdf/10.1103/zhx7-hcmm","primary_domain":"proteins-complexes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.1103/zhx7-hcmm","notes":"Final journal Table I, ESM2: OFS PP Aggregate Mean 0.403 checked directly; manuscript PMC11257618 printed the same value. Other models in the table are imported ProteinGym baselines; this row is the authors’ own evaluation."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-2ome-lm-2025","kind":"evaluation","name":"2OMe-LM: human RNA 2-prime-O-methylation site prediction","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["human RNA 2-prime-O-methylation site prediction"]},"source_ids":["2ome-lm-2025"],"links":[{"relation":"model","target_id":"reported-model-7f5b8234967c54"},{"relation":"benchmark","target_id":"reported-task-82fc7843f07324"},{"relation":"dataset","target_id":"reported-dataset-bd3d8e7d6cd196"}],"attributes":{"origin":"author_reported","protocol":"pretrained RNA language model predictor","version":"not stated in table","comparison":{"protocol_id":null,"dataset_version":null,"split":"5-fold cross-validation","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-antibody-deamidation-plm-2024","kind":"evaluation","name":"ESM-2 650M embeddings + classifier: antibody deamidation-site prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["antibody deamidation-site prediction"]},"source_ids":["antibody-deamidation-plm-2024"],"links":[{"relation":"model","target_id":"reported-model-4921459942b45f"},{"relation":"benchmark","target_id":"reported-task-0647b0364def8f"},{"relation":"dataset","target_id":"reported-dataset-0edd8f724db696"}],"attributes":{"origin":"author_reported","protocol":"global contextual embeddings only","version":"esm2_t33_650m_UR50D","comparison":{"protocol_id":null,"dataset_version":null,"split":"fivefold stratified CV","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-barcodebert-2026","kind":"evaluation","name":"BarcodeBERT (4–4-4): unseen-species genus classification","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["unseen-species genus classification"]},"source_ids":["barcodebert-2026"],"links":[{"relation":"model","target_id":"reported-model-05103f72325fe5"},{"relation":"benchmark","target_id":"reported-task-4a54ce01b5a855"},{"relation":"dataset","target_id":"reported-dataset-bc127dc9c441fe"}],"attributes":{"origin":"author_reported","protocol":"genus-level nearest-neighbor probe on species unseen in training","version":"4–4–4","comparison":{"protocol_id":null,"dataset_version":null,"split":"1-NN probe","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-birna-bert-2025","kind":"evaluation","name":"BiRNA-BERT: extremely long RNA species classification","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["extremely long RNA species classification"]},"source_ids":["birna-bert-2025"],"links":[{"relation":"model","target_id":"reported-model-d3fd83835a2d44"},{"relation":"benchmark","target_id":"reported-task-c40dac20d9af66"},{"relation":"dataset","target_id":"reported-dataset-ebc3f5fda43972"}],"attributes":{"origin":"author_reported","protocol":"adaptive tokenization on full-length long RNA sequences","version":"not stated in table","comparison":{"protocol_id":null,"dataset_version":null,"split":"paper evaluation","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-cathe2-2025","kind":"evaluation","name":"CATHe2 + ProstT5: CATH superfamily annotation","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["CATH superfamily annotation"]},"source_ids":["cathe2-2025"],"links":[{"relation":"model","target_id":"reported-model-49bc768f46b366"},{"relation":"benchmark","target_id":"reported-task-c98e91ffc7247d"},{"relation":"dataset","target_id":"reported-dataset-6e0c28dfde7337"}],"attributes":{"origin":"author_reported","protocol":"amino-acid and structural alphabet embedding classifier","version":"full ProstT5","comparison":{"protocol_id":null,"dataset_version":null,"split":"paper evaluation","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-clathrin-plm-2025","kind":"evaluation","name":"ESM-2 embedding + paper classifier: clathrin protein classification","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["clathrin protein classification"]},"source_ids":["clathrin-plm-2025"],"links":[{"relation":"model","target_id":"reported-model-e4710b1c3facf2"},{"relation":"benchmark","target_id":"reported-task-786c09824e9bf5"},{"relation":"dataset","target_id":"reported-dataset-0aab382ca2c063"}],"attributes":{"origin":"independent_paper","protocol":"single-feature ESM-2 embedding comparison","version":"not stated in table","comparison":{"protocol_id":null,"dataset_version":null,"split":"independent test","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-cobra-rna-binding-2026","kind":"evaluation","name":"ERNIE-RNA + CoBRA: RNA compound-binding site prediction","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA compound-binding site prediction"]},"source_ids":["cobra-rna-binding-2026"],"links":[{"relation":"model","target_id":"reported-model-7ad28cd57f5f5b"},{"relation":"benchmark","target_id":"reported-task-3a3bff34cce634"},{"relation":"dataset","target_id":"reported-dataset-b1af840b76b351"}],"attributes":{"origin":"author_reported","protocol":"ERNIE-RNA embedding with TCL focal loss","version":"not stated in table","comparison":{"protocol_id":null,"dataset_version":null,"split":"test set","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-codonbert-vaccines-2024","kind":"evaluation","name":"CodonBERT: flu-vaccine mRNA property prediction","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["flu-vaccine mRNA property prediction"]},"source_ids":["codonbert-vaccines-2024"],"links":[{"relation":"model","target_id":"reported-model-cd246741c378db"},{"relation":"benchmark","target_id":"reported-task-1c74661df2c401"},{"relation":"dataset","target_id":"reported-dataset-54b9bc432928d6"}],"attributes":{"origin":"author_reported","protocol":"codon-based model fine-tuned for downstream regression","version":"not stated in table","comparison":{"protocol_id":null,"dataset_version":null,"split":"paper evaluation","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-dart-eval-regulatory-2024","kind":"evaluation","name":"DNABERT-2: regulatory element identification","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["regulatory element identification"]},"source_ids":["dart-eval-regulatory-2024"],"links":[{"relation":"model","target_id":"reported-model-28413ae1766316"},{"relation":"benchmark","target_id":"reported-task-cdbee1c9285568"},{"relation":"dataset","target_id":"reported-dataset-b6ce37ba678d39"}],"attributes":{"origin":"independent_paper","protocol":"zero-shot likelihood ranking: higher likelihood for cCRE than matched control","version":"not stated in table","comparison":{"protocol_id":null,"dataset_version":null,"split":"paper evaluation","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-dnabert2-enhancer-2025","kind":"evaluation","name":"DNABERT2-Enhancer: enhancer recognition","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["enhancer recognition"]},"source_ids":["dnabert2-enhancer-2025"],"links":[{"relation":"model","target_id":"reported-model-d0e594ec3c0430"},{"relation":"benchmark","target_id":"reported-task-86a628af87ff8f"},{"relation":"dataset","target_id":"reported-dataset-6212e779949708"}],"attributes":{"origin":"author_reported","protocol":"first-layer enhancer versus non-enhancer classifier","version":"not stated in table","comparison":{"protocol_id":null,"dataset_version":null,"split":"5-fold cross-validation","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-eden-genomic-classification-2026","kind":"evaluation","name":"DNABERT-2: human core-promoter classification","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["human core-promoter classification"]},"source_ids":["eden-genomic-classification-2026"],"links":[{"relation":"model","target_id":"reported-model-2cb8118b4c77c0"},{"relation":"benchmark","target_id":"reported-task-9f62e739c6371e"},{"relation":"dataset","target_id":"reported-dataset-8e9488896becd4"}],"attributes":{"origin":"paper_compilation","protocol":"DNABERT-2 comparator in consolidated H-CPD table; rerun provenance not explicit","version":"not stated in table","comparison":{"protocol_id":null,"dataset_version":null,"split":"paper evaluation","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","original_evaluation":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-ernie-rna-2025","kind":"evaluation","name":"ERNIE-RNA: RNA secondary-structure prediction","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary-structure prediction"]},"source_ids":["ernie-rna-2025"],"links":[{"relation":"model","target_id":"reported-model-d023fbe78bc4df"},{"relation":"benchmark","target_id":"reported-task-a2bf7ddbc71d23"},{"relation":"dataset","target_id":"reported-dataset-abdfba8cce7486"}],"attributes":{"origin":"author_reported","protocol":"zero-shot attention-derived base-pair prediction","version":"86M","comparison":{"protocol_id":null,"dataset_version":null,"split":"cross-family test","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-esm2-ofs-fitness-2025","kind":"evaluation","name":"ESM2 OFS pseudo-perplexity: protein variant fitness prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["protein variant fitness prediction"]},"source_ids":["esm2-ofs-fitness-2025"],"links":[{"relation":"model","target_id":"reported-model-40ce004270dee4"},{"relation":"benchmark","target_id":"reported-task-c7a8a372f77886"},{"relation":"dataset","target_id":"reported-dataset-9c186c8f4ed3f4"}],"attributes":{"origin":"author_reported","protocol":"authors’ zero-shot ESM2 OFS pseudo-perplexity evaluation; aggregate mean across ProteinGym substitution assays","version":"not stated in table","comparison":{"protocol_id":null,"dataset_version":null,"split":"aggregate across assays","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-fusion-breakpoint-foundation-models-2026","kind":"evaluation","name":"Nucleotide Transformer + NN (middle): gene fusion breakpoint classification","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["gene fusion breakpoint classification"]},"source_ids":["fusion-breakpoint-foundation-models-2026"],"links":[{"relation":"model","target_id":"reported-model-1a67087ac262c5"},{"relation":"benchmark","target_id":"reported-task-ee34721cf55590"},{"relation":"dataset","target_id":"reported-dataset-f6922a9744ba27"}],"attributes":{"origin":"independent_paper","protocol":"middle embedding with neural-network classifier","version":"not stated in table","comparison":{"protocol_id":null,"dataset_version":null,"split":"full test set","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-genomic-tokenizer-selection-2025","kind":"evaluation","name":"Caduceus (character tokens): regulatory sequence classification","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["regulatory sequence classification"]},"source_ids":["genomic-tokenizer-selection-2025"],"links":[{"relation":"model","target_id":"reported-model-d5bc536ca6f0d3"},{"relation":"benchmark","target_id":"reported-task-cd127e56fb1f04"},{"relation":"dataset","target_id":"reported-dataset-0bba1c9a7ae410"}],"attributes":{"origin":"independent_paper","protocol":"task-category MCC across benchmark datasets","version":"3.9M parameter variant","comparison":{"protocol_id":null,"dataset_version":null,"split":"paper benchmark summary","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-gsmformer-ppi-2026","kind":"evaluation","name":"GSMFormer-PPI + ProstT5: protein-protein interaction prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["protein-protein interaction prediction"]},"source_ids":["gsmformer-ppi-2026"],"links":[{"relation":"model","target_id":"reported-model-73ae07fb5be204"},{"relation":"benchmark","target_id":"reported-task-dfa8f2285dbfa5"},{"relation":"dataset","target_id":"reported-dataset-07d355c146be1f"}],"attributes":{"origin":"author_reported","protocol":"ProstT5 embeddings as graph node features","version":"not stated in table","comparison":{"protocol_id":null,"dataset_version":null,"split":"test set","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-megsite-2025","kind":"evaluation","name":"MegSite + ESM3: DNA-binding residue prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["DNA-binding residue prediction"]},"source_ids":["megsite-2025"],"links":[{"relation":"model","target_id":"reported-model-86393c76dd8fa9"},{"relation":"benchmark","target_id":"reported-task-9917a0e69f33e7"},{"relation":"dataset","target_id":"reported-dataset-739aee3cf8d6f1"}],"attributes":{"origin":"author_reported","protocol":"ESM3 multimodal embedding ablation in MegSite","version":"not stated in table","comparison":{"protocol_id":null,"dataset_version":null,"split":"independent test","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-mrna-lm-2025","kind":"evaluation","name":"mRNA-LM: mRNA half-life prediction","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["mRNA half-life prediction"]},"source_ids":["mrna-lm-2025"],"links":[{"relation":"model","target_id":"reported-model-54d974e8e08043"},{"relation":"benchmark","target_id":"reported-task-5693847493f19f"},{"relation":"dataset","target_id":"reported-dataset-52f00ccaabf0d9"}],"attributes":{"origin":"author_reported","protocol":"average test performance across cross-validation splits","version":"not stated in table","comparison":{"protocol_id":null,"dataset_version":null,"split":"test set across CV splits","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-mrnabert-2025","kind":"evaluation","name":"mRNABERT: translation-efficiency prediction","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["translation-efficiency prediction"]},"source_ids":["mrnabert-2025"],"links":[{"relation":"model","target_id":"reported-model-13bd2a6c2d8178"},{"relation":"benchmark","target_id":"reported-task-f7142c3b3e0f3c"},{"relation":"dataset","target_id":"reported-dataset-1744719eef145b"}],"attributes":{"origin":"author_reported","protocol":"human translation-efficiency regression at 3066-nt input","version":"3066-nt input","comparison":{"protocol_id":null,"dataset_version":null,"split":"paper evaluation","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-mulan-2025","kind":"evaluation","name":"MULAN-ESM2 S: human protein-protein interaction prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["human protein-protein interaction prediction"]},"source_ids":["mulan-2025"],"links":[{"relation":"model","target_id":"reported-model-a29203c09857ef"},{"relation":"benchmark","target_id":"reported-task-6e54c7452b2b81"},{"relation":"dataset","target_id":"reported-dataset-38151fa548e291"}],"attributes":{"origin":"author_reported","protocol":"MULAN sequence-structure model based on ESM2 8M","version":"small ESM2 backbone","comparison":{"protocol_id":null,"dataset_version":null,"split":"paper evaluation","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-phylogpn-2025","kind":"evaluation","name":"PhyloGPN: ClinVar 3-prime UTR variant classification","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["ClinVar 3-prime UTR variant classification"]},"source_ids":["phylogpn-2025"],"links":[{"relation":"model","target_id":"reported-model-cbbe04b826ceff"},{"relation":"benchmark","target_id":"reported-task-ed3dd3b83c4505"},{"relation":"dataset","target_id":"reported-dataset-a28180d33f7a23"}],"attributes":{"origin":"author_reported","protocol":"log-likelihood-ratio scoring","version":"not stated in table","comparison":{"protocol_id":null,"dataset_version":null,"split":"paper evaluation","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-polya-glm-2025","kind":"evaluation","name":"HyenaDNA: polyadenylation site detection","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["polyadenylation site detection"]},"source_ids":["polya-glm-2025"],"links":[{"relation":"model","target_id":"reported-model-953007693fb72a"},{"relation":"benchmark","target_id":"reported-task-13dfe6b33e71ed"},{"relation":"dataset","target_id":"reported-dataset-55f200c9481409"}],"attributes":{"origin":"independent_paper","protocol":"few-shot Gene-Gene negative-set comparison","version":"not stated in table","comparison":{"protocol_id":null,"dataset_version":null,"split":"5-fold cross-validation","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-rlsite-rna-binding-2025","kind":"evaluation","name":"RLsite: RNA-small-molecule binding-site prediction","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA-small-molecule binding-site prediction"]},"source_ids":["rlsite-rna-binding-2025"],"links":[{"relation":"model","target_id":"reported-model-52b9c685d99290"},{"relation":"benchmark","target_id":"reported-task-b00a636d1ed8d9"},{"relation":"dataset","target_id":"reported-dataset-1c7f8ebb1968d9"}],"attributes":{"origin":"author_reported","protocol":"RNA language-model plus graph-attention classifier","version":"not stated in table","comparison":{"protocol_id":null,"dataset_version":null,"split":"paper evaluation","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-rnaret-2026","kind":"evaluation","name":"RNAret: miRNA-mRNA interaction prediction","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["miRNA-mRNA interaction prediction"]},"source_ids":["rnaret-2026"],"links":[{"relation":"model","target_id":"reported-model-7234658bc9c828"},{"relation":"benchmark","target_id":"reported-task-46e927bea10702"},{"relation":"dataset","target_id":"reported-dataset-99afd0c86b2954"}],"attributes":{"origin":"author_reported","protocol":"5-mer RNAret classifier; 72/8/20 train/validation/test split","version":"5-mer","comparison":{"protocol_id":null,"dataset_version":null,"split":"held-out test","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-spin-protein-function-2026","kind":"evaluation","name":"SPIN + ESM2-35M: protein function annotation","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["protein function annotation"]},"source_ids":["spin-protein-function-2026"],"links":[{"relation":"model","target_id":"reported-model-f8f0257b98749a"},{"relation":"benchmark","target_id":"reported-task-c4a578065f44b2"},{"relation":"dataset","target_id":"reported-dataset-dba1707164d296"}],"attributes":{"origin":"author_reported","protocol":"frozen ESM2-35M backbone in SPIN","version":"ESM2-35M frozen","comparison":{"protocol_id":null,"dataset_version":null,"split":"test set","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-structure-informed-plm-2025","kind":"evaluation","name":"structure-informed pLM: protein variant-effect classification","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["protein variant-effect classification"]},"source_ids":["structure-informed-plm-2025"],"links":[{"relation":"model","target_id":"reported-model-035a3ab36a3a6a"},{"relation":"benchmark","target_id":"reported-task-83be0998084c91"},{"relation":"dataset","target_id":"reported-dataset-2eaa2a051d45ee"}],"attributes":{"origin":"author_reported","protocol":"combined amino-acid, secondary structure, solvent accessibility and contact-map scoring","version":"not stated in table","comparison":{"protocol_id":null,"dataset_version":null,"split":"paper evaluation","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-001","kind":"evaluation","name":"Caduceus-Ph: Human 5mC detection","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["Human 5mC detection"]},"source_ids":["dna-foundation-models-2025"],"links":[{"relation":"model","target_id":"reported-model-47521865af7b04"},{"relation":"benchmark","target_id":"reported-task-988ff78f86471e"},{"relation":"dataset","target_id":"reported-dataset-463197d6a98b99"}],"attributes":{"origin":"independent_paper","protocol":"Binary epigenetic-modification classification as reported in the paper.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-002","kind":"evaluation","name":"NT-v2: Human 5mC detection","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["Human 5mC detection"]},"source_ids":["dna-foundation-models-2025"],"links":[{"relation":"model","target_id":"reported-model-3af86cb274f658"},{"relation":"benchmark","target_id":"reported-task-988ff78f86471e"},{"relation":"dataset","target_id":"reported-dataset-463197d6a98b99"}],"attributes":{"origin":"independent_paper","protocol":"Binary epigenetic-modification classification as reported in the paper.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-003","kind":"evaluation","name":"ENBED: Enhancer classification","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["Enhancer classification"]},"source_ids":["enbed-2024"],"links":[{"relation":"model","target_id":"reported-model-8db190bee6aae5"},{"relation":"benchmark","target_id":"reported-task-132da895d4c381"},{"relation":"dataset","target_id":"reported-dataset-f0bf60a62ad7c3"}],"attributes":{"origin":"author_reported","protocol":"Reported Genomic Benchmarks classification accuracy.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-004","kind":"evaluation","name":"ENBED (GRCh38): Enhancer classification","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["Enhancer classification"]},"source_ids":["enbed-2024"],"links":[{"relation":"model","target_id":"reported-model-bdb1db16d3389d"},{"relation":"benchmark","target_id":"reported-task-132da895d4c381"},{"relation":"dataset","target_id":"reported-dataset-f0bf60a62ad7c3"}],"attributes":{"origin":"author_reported","protocol":"ENBED trained on GRCh38; reported Genomic Benchmarks classification accuracy.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-005","kind":"evaluation","name":"DNABERT-2: G-quadruplex classification","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["G-quadruplex classification"]},"source_ids":["quadruplex-llm-benchmark-2025"],"links":[{"relation":"model","target_id":"reported-model-ade36035f58f27"},{"relation":"benchmark","target_id":"reported-task-c9d2a6435979e9"},{"relation":"dataset","target_id":"reported-dataset-9e9d18bc5bfb8b"}],"attributes":{"origin":"independent_paper","protocol":"Pretrained model evaluated on KEx as reported in Table 5.","version":"117M","comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-006","kind":"evaluation","name":"Caduceus: G-quadruplex classification","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["G-quadruplex classification"]},"source_ids":["quadruplex-llm-benchmark-2025"],"links":[{"relation":"model","target_id":"reported-model-abc19288fe9009"},{"relation":"benchmark","target_id":"reported-task-c9d2a6435979e9"},{"relation":"dataset","target_id":"reported-dataset-9e9d18bc5bfb8b"}],"attributes":{"origin":"independent_paper","protocol":"Pretrained model evaluated on KEx as reported in Table 5.","version":"8M","comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-007","kind":"evaluation","name":"HyenaDNA: Enhancer-target gene prediction","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["Enhancer-target gene prediction"]},"source_ids":["dnalongbench-2025"],"links":[{"relation":"model","target_id":"reported-model-9b3bc255532dd3"},{"relation":"benchmark","target_id":"reported-task-2cbac97dd849f5"},{"relation":"dataset","target_id":"reported-dataset-fcb5752d916d5d"}],"attributes":{"origin":"independent_paper","protocol":"Long-range ETGP benchmark; source table reports AUROC.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-008","kind":"evaluation","name":"Caduceus-Ph: Enhancer-target gene prediction","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["Enhancer-target gene prediction"]},"source_ids":["dnalongbench-2025"],"links":[{"relation":"model","target_id":"reported-model-a7cfacf25d97ad"},{"relation":"benchmark","target_id":"reported-task-2cbac97dd849f5"},{"relation":"dataset","target_id":"reported-dataset-fcb5752d916d5d"}],"attributes":{"origin":"independent_paper","protocol":"Long-range ETGP benchmark; source table reports AUROC.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-009","kind":"evaluation","name":"RiNALMo: Mean ribosome load from MPRA","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["Mean ribosome load from MPRA"]},"source_ids":["mrnabench-2025"],"links":[{"relation":"model","target_id":"reported-model-3e58d0faf88d2e"},{"relation":"benchmark","target_id":"reported-task-57dc3dcdb67a81"},{"relation":"dataset","target_id":"reported-dataset-3a3e3880a3fed0"}],"attributes":{"origin":"independent_paper","protocol":"Linear probe; mean across ten random seeds.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-010","kind":"evaluation","name":"RNA-FM: Mean ribosome load from MPRA","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["Mean ribosome load from MPRA"]},"source_ids":["mrnabench-2025"],"links":[{"relation":"model","target_id":"reported-model-43cf51abca83d1"},{"relation":"benchmark","target_id":"reported-task-57dc3dcdb67a81"},{"relation":"dataset","target_id":"reported-dataset-3a3e3880a3fed0"}],"attributes":{"origin":"independent_paper","protocol":"Linear probe; mean across ten random seeds.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-011","kind":"evaluation","name":"BPfold: RNA secondary structure","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary structure"]},"source_ids":["bpfold-2025"],"links":[{"relation":"model","target_id":"reported-model-c464bface507ee"},{"relation":"benchmark","target_id":"reported-task-dc82fcbfb44935"},{"relation":"dataset","target_id":"reported-dataset-8317793f18b026"}],"attributes":{"origin":"author_reported","protocol":"Family-wise evaluation of canonical base-pair predictions.","version":null,"comparison":{"protocol_id":null,"dataset_version":"116 RNAs","split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-012","kind":"evaluation","name":"RNAfold: RNA secondary structure","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary structure"]},"source_ids":["bpfold-2025"],"links":[{"relation":"model","target_id":"reported-model-52eee4cc67ca26"},{"relation":"benchmark","target_id":"reported-task-dc82fcbfb44935"},{"relation":"dataset","target_id":"reported-dataset-8317793f18b026"}],"attributes":{"origin":"independent_paper","protocol":"Family-wise evaluation of canonical base-pair predictions.","version":null,"comparison":{"protocol_id":null,"dataset_version":"116 RNAs","split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-013","kind":"evaluation","name":"TU-Fold (aug): RNA secondary structure","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary structure"]},"source_ids":["tu-fold-2025"],"links":[{"relation":"model","target_id":"reported-model-d1cd9a425f9bbd"},{"relation":"benchmark","target_id":"reported-task-5ec7581b246ea6"},{"relation":"dataset","target_id":"reported-dataset-f2e729f333a333"}],"attributes":{"origin":"author_reported","protocol":"Three-fold training and evaluation; source reports mean and standard deviation.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-014","kind":"evaluation","name":"UFold: RNA secondary structure","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary structure"]},"source_ids":["tu-fold-2025"],"links":[{"relation":"model","target_id":"reported-model-e3abb0b9a2ec79"},{"relation":"benchmark","target_id":"reported-task-5ec7581b246ea6"},{"relation":"dataset","target_id":"reported-dataset-f2e729f333a333"}],"attributes":{"origin":"independent_paper","protocol":"Three-fold training and evaluation; source reports mean and standard deviation.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-015","kind":"evaluation","name":"DEBFold: RNA secondary structure","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary structure"]},"source_ids":["debfold-2024"],"links":[{"relation":"model","target_id":"reported-model-3f850c08d76410"},{"relation":"benchmark","target_id":"reported-task-016f70615f2cfc"},{"relation":"dataset","target_id":"reported-dataset-18ebde58579c2b"}],"attributes":{"origin":"author_reported","protocol":"Median F1 on the prepared TestSetβ.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-016","kind":"evaluation","name":"RNAfold: RNA secondary structure","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary structure"]},"source_ids":["debfold-2024"],"links":[{"relation":"model","target_id":"reported-model-ed7f0db85facb1"},{"relation":"benchmark","target_id":"reported-task-016f70615f2cfc"},{"relation":"dataset","target_id":"reported-dataset-18ebde58579c2b"}],"attributes":{"origin":"independent_paper","protocol":"Median F1 on the prepared TestSetβ.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-017","kind":"evaluation","name":"ESM-2: Zero-shot substitution mutation effects: stability","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["Zero-shot substitution mutation effects: stability"]},"source_ids":["proteingym-2023"],"links":[{"relation":"model","target_id":"reported-model-d326e3c4e3ba20"},{"relation":"benchmark","target_id":"reported-task-6243658a1bc215"},{"relation":"dataset","target_id":"reported-dataset-7cec655cd742f3"}],"attributes":{"origin":"independent_paper","protocol":"Zero-shot mutation scores; average Spearman across stability-category assays.","version":"15B","comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-018","kind":"evaluation","name":"ProteinMPNN: Zero-shot substitution mutation effects: stability","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["Zero-shot substitution mutation effects: stability"]},"source_ids":["proteingym-2023"],"links":[{"relation":"model","target_id":"reported-model-e0443048c6e110"},{"relation":"benchmark","target_id":"reported-task-6243658a1bc215"},{"relation":"dataset","target_id":"reported-dataset-7cec655cd742f3"}],"attributes":{"origin":"independent_paper","protocol":"Zero-shot mutation scores; average Spearman across stability-category assays.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-019","kind":"evaluation","name":"FUJISAN: Enzyme functional identity prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["Enzyme functional identity prediction"]},"source_ids":["fujisan-2024"],"links":[{"relation":"model","target_id":"reported-model-9c10fbec02a365"},{"relation":"benchmark","target_id":"reported-task-1ebf9b408517f9"},{"relation":"dataset","target_id":"reported-dataset-5197cca532f89d"}],"attributes":{"origin":"author_reported","protocol":"Sequence and structural feature integration; paper-reported test sub-dataset.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-020","kind":"evaluation","name":"ESM2: Enzyme functional identity prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["Enzyme functional identity prediction"]},"source_ids":["fujisan-2024"],"links":[{"relation":"model","target_id":"reported-model-ccd1160ad4ec27"},{"relation":"benchmark","target_id":"reported-task-1ebf9b408517f9"},{"relation":"dataset","target_id":"reported-dataset-5197cca532f89d"}],"attributes":{"origin":"independent_paper","protocol":"Comparator evaluated on the paper-reported test sub-dataset.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-021","kind":"evaluation","name":"ESM-2: Mutated RBD binding prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["Mutated RBD binding prediction"]},"source_ids":["prime-2026"],"links":[{"relation":"model","target_id":"reported-model-d0d5df2beb02b2"},{"relation":"benchmark","target_id":"reported-task-00e594df6a182d"},{"relation":"dataset","target_id":"reported-dataset-becc215358afd0"}],"attributes":{"origin":"independent_paper","protocol":"Frozen mean-pooled representation with downstream regression; position-stratified split.","version":"8M","comparison":{"protocol_id":null,"dataset_version":null,"split":"position-stratified","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-022","kind":"evaluation","name":"ESM-C: Mutated RBD binding prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["Mutated RBD binding prediction"]},"source_ids":["prime-2026"],"links":[{"relation":"model","target_id":"reported-model-f83c0b833411a7"},{"relation":"benchmark","target_id":"reported-task-00e594df6a182d"},{"relation":"dataset","target_id":"reported-dataset-becc215358afd0"}],"attributes":{"origin":"independent_paper","protocol":"Frozen mean-pooled representation with downstream regression; position-stratified split.","version":"300M","comparison":{"protocol_id":null,"dataset_version":null,"split":"position-stratified","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-023","kind":"evaluation","name":"PST: Zero-shot variant effect prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["Zero-shot variant effect prediction"]},"source_ids":["pst-2025"],"links":[{"relation":"model","target_id":"reported-model-7dd5992188a868"},{"relation":"benchmark","target_id":"reported-task-a5141363b0ee45"},{"relation":"dataset","target_id":"reported-dataset-bd9255afb783d6"}],"attributes":{"origin":"author_reported","protocol":"Zero-shot VEP; paper averages absolute Spearman correlations.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-024","kind":"evaluation","name":"ESM-2: Zero-shot variant effect prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["Zero-shot variant effect prediction"]},"source_ids":["pst-2025"],"links":[{"relation":"model","target_id":"reported-model-d25dab1a9c4fff"},{"relation":"benchmark","target_id":"reported-task-a5141363b0ee45"},{"relation":"dataset","target_id":"reported-dataset-bd9255afb783d6"}],"attributes":{"origin":"independent_paper","protocol":"Zero-shot VEP; paper averages absolute Spearman correlations.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-025","kind":"evaluation","name":"scGPT: Cell-type identification","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Cell-type identification"]},"source_ids":["single-cell-peft-2024"],"links":[{"relation":"model","target_id":"reported-model-3bdb3093e8d531"},{"relation":"benchmark","target_id":"reported-task-5b929593eefc76"},{"relation":"dataset","target_id":"reported-dataset-488d5de6bb9c1b"}],"attributes":{"origin":"independent_paper","protocol":"Native scLLM cell-type identification as reported in Table 2.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-026","kind":"evaluation","name":"Geneformer: Cell-type identification","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Cell-type identification"]},"source_ids":["single-cell-peft-2024"],"links":[{"relation":"model","target_id":"reported-model-b46ae14b9927ac"},{"relation":"benchmark","target_id":"reported-task-5b929593eefc76"},{"relation":"dataset","target_id":"reported-dataset-488d5de6bb9c1b"}],"attributes":{"origin":"independent_paper","protocol":"Native scLLM cell-type identification as reported in Table 2.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-027","kind":"evaluation","name":"C2S (GPT-2 Large): Combinatorial cell-label classification","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Combinatorial cell-label classification"]},"source_ids":["cell2sentence-2024"],"links":[{"relation":"model","target_id":"reported-model-ab02228f50a37c"},{"relation":"benchmark","target_id":"reported-task-7efe245cc94ee5"},{"relation":"dataset","target_id":"reported-dataset-87e91d9d6e6f4b"}],"attributes":{"origin":"author_reported","protocol":"Partial-credit labels including cell type, perturbation, and dose.","version":"GPT-2 Large","comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-028","kind":"evaluation","name":"Geneformer: Combinatorial cell-label classification","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Combinatorial cell-label classification"]},"source_ids":["cell2sentence-2024"],"links":[{"relation":"model","target_id":"reported-model-06816ce9073144"},{"relation":"benchmark","target_id":"reported-task-7efe245cc94ee5"},{"relation":"dataset","target_id":"reported-dataset-87e91d9d6e6f4b"}],"attributes":{"origin":"independent_paper","protocol":"Partial-credit labels including cell type, perturbation, and dose.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-029","kind":"evaluation","name":"scGPT: Cell-type annotation","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Cell-type annotation"]},"source_ids":["scelmo-2025"],"links":[{"relation":"model","target_id":"reported-model-77ad27d4098177"},{"relation":"benchmark","target_id":"reported-task-660753ec94e631"},{"relation":"dataset","target_id":"reported-dataset-8f123f006964ad"}],"attributes":{"origin":"paper_compilation","protocol":"Zero-shot setting; source caption says some comparator rows come from GenePT.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","original_evaluation":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-030","kind":"evaluation","name":"Geneformer: Cell-type annotation","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Cell-type annotation"]},"source_ids":["scelmo-2025"],"links":[{"relation":"model","target_id":"reported-model-d60f505aabb19c"},{"relation":"benchmark","target_id":"reported-task-660753ec94e631"},{"relation":"dataset","target_id":"reported-dataset-8f123f006964ad"}],"attributes":{"origin":"paper_compilation","protocol":"Zero-shot setting; source caption says some comparator rows come from GenePT.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","original_evaluation":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-031","kind":"evaluation","name":"scRegNet (Geneformer backbone): Gene-regulatory link prediction","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Gene-regulatory link prediction"]},"source_ids":["scregnet-2025"],"links":[{"relation":"model","target_id":"reported-model-60455ff7cc0c15"},{"relation":"benchmark","target_id":"reported-task-3063ed4da76b4b"},{"relation":"dataset","target_id":"reported-dataset-2ad2fad5e1cd0a"}],"attributes":{"origin":"author_reported","protocol":"TFs plus 500 variable genes; mean from 50 independent evaluations.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-032","kind":"evaluation","name":"scRegNet (scBERT backbone): Gene-regulatory link prediction","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Gene-regulatory link prediction"]},"source_ids":["scregnet-2025"],"links":[{"relation":"model","target_id":"reported-model-89f5a8f309fa18"},{"relation":"benchmark","target_id":"reported-task-3063ed4da76b4b"},{"relation":"dataset","target_id":"reported-dataset-2ad2fad5e1cd0a"}],"attributes":{"origin":"author_reported","protocol":"TFs plus 500 variable genes; mean from 50 independent evaluations.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-033","kind":"evaluation","name":"ProkBERT-mini: E. coli sigma70 promoter prediction","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["E. coli sigma70 promoter prediction"]},"source_ids":["prokbert-2024"],"links":[{"relation":"model","target_id":"reported-model-4438513d9cd42c"},{"relation":"benchmark","target_id":"reported-task-3891811dcce8b3"},{"relation":"dataset","target_id":"reported-dataset-48def1da574597"}],"attributes":{"origin":"author_reported","protocol":"Promoter versus non-promoter classification.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-034","kind":"evaluation","name":"Promotech: E. coli sigma70 promoter prediction","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["E. coli sigma70 promoter prediction"]},"source_ids":["prokbert-2024"],"links":[{"relation":"model","target_id":"reported-model-0d147487bf97be"},{"relation":"benchmark","target_id":"reported-task-3891811dcce8b3"},{"relation":"dataset","target_id":"reported-dataset-48def1da574597"}],"attributes":{"origin":"independent_paper","protocol":"Promoter versus non-promoter classification.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-035","kind":"evaluation","name":"Eco70PromBERT: E. coli sigma70 promoter prediction","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["E. coli sigma70 promoter prediction"]},"source_ids":["cyaprombert-2022"],"links":[{"relation":"model","target_id":"reported-model-5b70fccb70bb70"},{"relation":"benchmark","target_id":"reported-task-e5c34f686ac403"},{"relation":"dataset","target_id":"reported-dataset-a1da4a37eb46a5"}],"attributes":{"origin":"author_reported","protocol":"BERT-base with 1bp tokenizer; 110 promoters and 108 non-promoters.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":"independent test","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-036","kind":"evaluation","name":"iPro70-FMWin: E. coli sigma70 promoter prediction","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["E. coli sigma70 promoter prediction"]},"source_ids":["cyaprombert-2022"],"links":[{"relation":"model","target_id":"reported-model-23cb15b93c00ff"},{"relation":"benchmark","target_id":"reported-task-e5c34f686ac403"},{"relation":"dataset","target_id":"reported-dataset-a1da4a37eb46a5"}],"attributes":{"origin":"independent_paper","protocol":"Compared on the same independent test dataset; 110 promoters and 108 non-promoters.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":"independent test","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-037","kind":"evaluation","name":"EVO2: Genome-wide prophage detection","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Genome-wide prophage detection"]},"source_ids":["lambda-prophage-2026"],"links":[{"relation":"model","target_id":"reported-model-aa763db2cfdeff"},{"relation":"benchmark","target_id":"reported-task-dd001540e0f4ec"},{"relation":"dataset","target_id":"reported-dataset-1b4f6ea24c0587"}],"attributes":{"origin":"independent_paper","protocol":"Genomic language model fine-tuned for prophage detection; genome-wide evaluation.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-038","kind":"evaluation","name":"geNomad: Genome-wide prophage detection","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Genome-wide prophage detection"]},"source_ids":["lambda-prophage-2026"],"links":[{"relation":"model","target_id":"reported-model-d0677d52d2b9fd"},{"relation":"benchmark","target_id":"reported-task-dd001540e0f4ec"},{"relation":"dataset","target_id":"reported-dataset-1b4f6ea24c0587"}],"attributes":{"origin":"independent_paper","protocol":"Traditional specialist comparator; genome-wide evaluation.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-039","kind":"evaluation","name":"NABAS+: Metagenomic taxonomic classification","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Metagenomic taxonomic classification"]},"source_ids":["nabas-plus-2025"],"links":[{"relation":"model","target_id":"reported-model-e7d203bd99ca99"},{"relation":"benchmark","target_id":"reported-task-92137759a9e7b0"},{"relation":"dataset","target_id":"reported-dataset-b462aa24561fba"}],"attributes":{"origin":"author_reported","protocol":"Newly generated sample19 used for classifier comparison.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-040","kind":"evaluation","name":"MetaPhlAn3: Metagenomic taxonomic classification","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Metagenomic taxonomic classification"]},"source_ids":["nabas-plus-2025"],"links":[{"relation":"model","target_id":"reported-model-df4084611520b7"},{"relation":"benchmark","target_id":"reported-task-92137759a9e7b0"},{"relation":"dataset","target_id":"reported-dataset-b462aa24561fba"}],"attributes":{"origin":"independent_paper","protocol":"Newly generated sample19 used for classifier comparison.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-041","kind":"evaluation","name":"Chai-1: Lipid–protein binding pose","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Lipid–protein binding pose"]},"source_ids":["lipp-2026"],"links":[{"relation":"model","target_id":"reported-model-eae60780097101"},{"relation":"benchmark","target_id":"reported-task-ff2dec63c5a3dd"},{"relation":"dataset","target_id":"reported-dataset-6a44f5946cd7ab"}],"attributes":{"origin":"independent_paper","protocol":"Top-scoring pose; all-atom lipid RMSD below 2 Å.","version":null,"comparison":{"protocol_id":null,"dataset_version":"331 complexes","split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-042","kind":"evaluation","name":"DiffDock-L: Lipid–protein binding pose","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Lipid–protein binding pose"]},"source_ids":["lipp-2026"],"links":[{"relation":"model","target_id":"reported-model-51ed86132346a0"},{"relation":"benchmark","target_id":"reported-task-ff2dec63c5a3dd"},{"relation":"dataset","target_id":"reported-dataset-6a44f5946cd7ab"}],"attributes":{"origin":"independent_paper","protocol":"Top-scoring pose; all-atom lipid RMSD below 2 Å.","version":null,"comparison":{"protocol_id":null,"dataset_version":"331 complexes","split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-043","kind":"evaluation","name":"DiffDock-NMDN: Protein–ligand virtual screening","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand virtual screening"]},"source_ids":["nmdn-2025"],"links":[{"relation":"model","target_id":"reported-model-6c0bc8d297cc7a"},{"relation":"benchmark","target_id":"reported-task-a7803ecf7708cc"},{"relation":"dataset","target_id":"reported-dataset-065b9fcc8da573"}],"attributes":{"origin":"author_reported","protocol":"NMDN scoring on DiffDock-NMDN blind docked poses; not ligand-pose RMSD.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-044","kind":"evaluation","name":"Vina: Protein–ligand virtual screening","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand virtual screening"]},"source_ids":["nmdn-2025"],"links":[{"relation":"model","target_id":"reported-model-1e51ccbfd2de61"},{"relation":"benchmark","target_id":"reported-task-a7803ecf7708cc"},{"relation":"dataset","target_id":"reported-dataset-065b9fcc8da573"}],"attributes":{"origin":"independent_paper","protocol":"Vina scoring on the same DiffDock-NMDN blind docked poses; not ligand-pose RMSD.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-045","kind":"evaluation","name":"Boltz-1: Protein–ligand pose prediction","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand pose prediction"]},"source_ids":["boltz-stereochemistry-2025"],"links":[{"relation":"model","target_id":"reported-model-d9a06805b36b8a"},{"relation":"benchmark","target_id":"reported-task-bf513ed6db92c5"},{"relation":"dataset","target_id":"reported-dataset-5afaefb87c8a94"}],"attributes":{"origin":"independent_paper","protocol":"All entries; authors note this dataset contains structures seen during model training.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-046","kind":"evaluation","name":"DiffDock: Protein–ligand pose prediction","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand pose prediction"]},"source_ids":["boltz-stereochemistry-2025"],"links":[{"relation":"model","target_id":"reported-model-7f6ffd9e2a08be"},{"relation":"benchmark","target_id":"reported-task-bf513ed6db92c5"},{"relation":"dataset","target_id":"reported-dataset-5afaefb87c8a94"}],"attributes":{"origin":"independent_paper","protocol":"All entries; rigid-protein docking comparator; authors note this dataset contains structures seen during model training.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-047","kind":"evaluation","name":"Boltz-2: Ligand potency prediction using generated poses","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Ligand potency prediction using generated poses"]},"source_ids":["mpro-pose-affinity-2025"],"links":[{"relation":"model","target_id":"reported-model-cdc9aabf4efc04"},{"relation":"benchmark","target_id":"reported-task-d5f897ab0f6f67"},{"relation":"dataset","target_id":"reported-dataset-235520c84b737f"}],"attributes":{"origin":"independent_paper","protocol":"Potency prediction using Boltz-2 ligand-pose generation protocol; see paper scoring pipeline.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-048","kind":"evaluation","name":"DiffDock: Ligand potency prediction using generated poses","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Ligand potency prediction using generated poses"]},"source_ids":["mpro-pose-affinity-2025"],"links":[{"relation":"model","target_id":"reported-model-415ee22f46526c"},{"relation":"benchmark","target_id":"reported-task-d5f897ab0f6f67"},{"relation":"dataset","target_id":"reported-dataset-235520c84b737f"}],"attributes":{"origin":"independent_paper","protocol":"Potency prediction using DiffDock ligand-pose generation plus paper scoring pipeline; not a native DiffDock affinity score.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-003","kind":"evaluation","name":"Mouse-Geneformer: Human thymus cell-type classification","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Human thymus cell-type classification"]},"source_ids":["mouse-geneformer-2025"],"links":[{"relation":"model","target_id":"reported-model-e2f2f0d4830bb0"},{"relation":"benchmark","target_id":"reported-task-031186b57c62de"},{"relation":"dataset","target_id":"reported-dataset-477a9082515406"}],"attributes":{"origin":"author_reported","protocol":"Ortholog-based gene conversion; zero-shot mouse model on human cells.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-004","kind":"evaluation","name":"Human-Geneformer: Human thymus cell-type classification","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Human thymus cell-type classification"]},"source_ids":["mouse-geneformer-2025"],"links":[{"relation":"model","target_id":"reported-model-10d85f2a035720"},{"relation":"benchmark","target_id":"reported-task-031186b57c62de"},{"relation":"dataset","target_id":"reported-dataset-477a9082515406"}],"attributes":{"origin":"independent_paper","protocol":"Native human model; zero-shot setting.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-005","kind":"evaluation","name":"scLLMDA: Cross-platform scATAC cell-type annotation","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Cross-platform scATAC cell-type annotation"]},"source_ids":["scatac-llmda-2026"],"links":[{"relation":"model","target_id":"reported-model-a0db32ae53e5ed"},{"relation":"benchmark","target_id":"reported-task-d82b6284f3f431"},{"relation":"dataset","target_id":"reported-dataset-7fc59ce4c0ceaa"}],"attributes":{"origin":"author_reported","protocol":"Cross-platform reference-query cell-type annotation.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-006","kind":"evaluation","name":"MINGLE: Cross-platform scATAC cell-type annotation","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Cross-platform scATAC cell-type annotation"]},"source_ids":["scatac-llmda-2026"],"links":[{"relation":"model","target_id":"reported-model-f0c630d0565e64"},{"relation":"benchmark","target_id":"reported-task-d82b6284f3f431"},{"relation":"dataset","target_id":"reported-dataset-7fc59ce4c0ceaa"}],"attributes":{"origin":"independent_paper","protocol":"Cross-platform reference-query comparator.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-011","kind":"evaluation","name":"GenePT-w: Cell-type structure in frozen embeddings","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Cell-type structure in frozen embeddings"]},"source_ids":["genept-2024"],"links":[{"relation":"model","target_id":"reported-model-7c595040de69bc"},{"relation":"benchmark","target_id":"reported-task-4df1fb456d3deb"},{"relation":"dataset","target_id":"reported-dataset-eaa2965545c87b"}],"attributes":{"origin":"author_reported","protocol":"k-means on pretrained cell embeddings; agreement with original cell-type labels.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-012","kind":"evaluation","name":"scGPT: Cell-type structure in frozen embeddings","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Cell-type structure in frozen embeddings"]},"source_ids":["genept-2024"],"links":[{"relation":"model","target_id":"reported-model-399b1ce87a3f6d"},{"relation":"benchmark","target_id":"reported-task-4df1fb456d3deb"},{"relation":"dataset","target_id":"reported-dataset-eaa2965545c87b"}],"attributes":{"origin":"independent_paper","protocol":"k-means on pretrained cell embeddings; agreement with original cell-type labels.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-013","kind":"evaluation","name":"Best frozen single-cell foundation model: Donor-aware age-class prediction","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Donor-aware age-class prediction"]},"source_ids":["single-cell-aging-probes-2026"],"links":[{"relation":"model","target_id":"reported-model-148b613975b6eb"},{"relation":"benchmark","target_id":"reported-task-d6018ca598e525"},{"relation":"dataset","target_id":"reported-dataset-571ce000cd74b9"}],"attributes":{"origin":"independent_paper","protocol":"Same donor-aware splits and logistic-regression probe as expression PCA; text names Geneformer as best model on AIDA v2.","version":null,"comparison":{"protocol_id":null,"dataset_version":"622 donors","split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-014","kind":"evaluation","name":"Gene-expression PCA: Donor-aware age-class prediction","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Donor-aware age-class prediction"]},"source_ids":["single-cell-aging-probes-2026"],"links":[{"relation":"model","target_id":"reported-model-177f32ce8189a0"},{"relation":"benchmark","target_id":"reported-task-d6018ca598e525"},{"relation":"dataset","target_id":"reported-dataset-571ce000cd74b9"}],"attributes":{"origin":"independent_paper","protocol":"Fifty-component gene-expression PCA with the same donor-aware probe splits.","version":null,"comparison":{"protocol_id":null,"dataset_version":"622 donors","split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-015","kind":"evaluation","name":"scaLR: PBMC cell-type classification","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["PBMC cell-type classification"]},"source_ids":["scalr-2025"],"links":[{"relation":"model","target_id":"reported-model-7cf2f9951e1dba"},{"relation":"benchmark","target_id":"reported-task-b46b7b839bff93"},{"relation":"dataset","target_id":"reported-dataset-33a41fe5fc66cf"}],"attributes":{"origin":"author_reported","protocol":"All features and samples from PBMCs-BS.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-016","kind":"evaluation","name":"scVI + scANVI: PBMC cell-type classification","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["PBMC cell-type classification"]},"source_ids":["scalr-2025"],"links":[{"relation":"model","target_id":"reported-model-1be5c4b7c52a41"},{"relation":"benchmark","target_id":"reported-task-b46b7b839bff93"},{"relation":"dataset","target_id":"reported-dataset-33a41fe5fc66cf"}],"attributes":{"origin":"independent_paper","protocol":"All features and samples from PBMCs-BS; comparison pipeline combines scVI and scANVI.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-017","kind":"evaluation","name":"scXDR: Cross-dataset single-cell drug response transfer","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Cross-dataset single-cell drug response transfer"]},"source_ids":["scxdr-2026"],"links":[{"relation":"model","target_id":"reported-model-9f39de53f7a139"},{"relation":"benchmark","target_id":"reported-task-167f08013c270e"},{"relation":"dataset","target_id":"reported-dataset-f7210686a78474"}],"attributes":{"origin":"author_reported","protocol":"Single-cell-to-single-cell transfer; source scenario 2.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-018","kind":"evaluation","name":"scVI: Cross-dataset single-cell drug response transfer","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Cross-dataset single-cell drug response transfer"]},"source_ids":["scxdr-2026"],"links":[{"relation":"model","target_id":"reported-model-f23306b94dc7b6"},{"relation":"benchmark","target_id":"reported-task-167f08013c270e"},{"relation":"dataset","target_id":"reported-dataset-f7210686a78474"}],"attributes":{"origin":"independent_paper","protocol":"Single-cell-to-single-cell transfer; source scenario 2.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-019","kind":"evaluation","name":"CAMMiQ: Strain-level abundance quantification","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Strain-level abundance quantification"]},"source_ids":["cammiq-2022"],"links":[{"relation":"model","target_id":"reported-model-32a19f43a4c254"},{"relation":"benchmark","target_id":"reported-task-571f0a2e7faed3"},{"relation":"dataset","target_id":"reported-dataset-72e837e5b97041"}],"attributes":{"origin":"author_reported","protocol":"Strain-level quantification on the HumanGut-all synthetic query.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-020","kind":"evaluation","name":"Kraken2: Strain-level abundance quantification","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Strain-level abundance quantification"]},"source_ids":["cammiq-2022"],"links":[{"relation":"model","target_id":"reported-model-70f57ebb163a5c"},{"relation":"benchmark","target_id":"reported-task-571f0a2e7faed3"},{"relation":"dataset","target_id":"reported-dataset-72e837e5b97041"}],"attributes":{"origin":"independent_paper","protocol":"Strain-level quantification on the HumanGut-all synthetic query.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-021","kind":"evaluation","name":"Lazypipe-nt: Simulated metagenome virus-taxon retrieval","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Simulated metagenome virus-taxon retrieval"]},"source_ids":["lazypipe-2020"],"links":[{"relation":"model","target_id":"reported-model-8cb3dd4e9f5b10"},{"relation":"benchmark","target_id":"reported-task-369dcfef14c4a9"},{"relation":"dataset","target_id":"reported-dataset-450c1af18cc623"}],"attributes":{"origin":"author_reported","protocol":"Genus-rank viral taxon retrieval.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-022","kind":"evaluation","name":"Kraken2: Simulated metagenome virus-taxon retrieval","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Simulated metagenome virus-taxon retrieval"]},"source_ids":["lazypipe-2020"],"links":[{"relation":"model","target_id":"reported-model-673b8f46361000"},{"relation":"benchmark","target_id":"reported-task-369dcfef14c4a9"},{"relation":"dataset","target_id":"reported-dataset-450c1af18cc623"}],"attributes":{"origin":"independent_paper","protocol":"Genus-rank viral taxon retrieval.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-023","kind":"evaluation","name":"NCD-gzip: CAMI II superkingdom read classification","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["CAMI II superkingdom read classification"]},"source_ids":["ncd-metagenomics-2026"],"links":[{"relation":"model","target_id":"reported-model-7b052acf17b5ba"},{"relation":"benchmark","target_id":"reported-task-a2c37b8c420bc3"},{"relation":"dataset","target_id":"reported-dataset-beb4f5da29da0a"}],"attributes":{"origin":"author_reported","protocol":"Superkingdom-level macro-averaged F1; NCD assigns every read.","version":null,"comparison":{"protocol_id":null,"dataset_version":"10,000 reads","split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-024","kind":"evaluation","name":"NCD-gzip: CAMI II phylum read classification","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["CAMI II phylum read classification"]},"source_ids":["ncd-metagenomics-2026"],"links":[{"relation":"model","target_id":"reported-model-7b052acf17b5ba"},{"relation":"benchmark","target_id":"reported-task-45105e1c486251"},{"relation":"dataset","target_id":"reported-dataset-beb4f5da29da0a"}],"attributes":{"origin":"author_reported","protocol":"Phylum-level macro-averaged F1; distinct taxonomic rank from the other row.","version":null,"comparison":{"protocol_id":null,"dataset_version":"10,000 reads","split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-025","kind":"evaluation","name":"VIBRANT: Simulated prophage-contig detection","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Simulated prophage-contig detection"]},"source_ids":["viral-contig-simulation-2021"],"links":[{"relation":"model","target_id":"reported-model-27dca28a87cf3c"},{"relation":"benchmark","target_id":"reported-task-53e3d216eef6db"},{"relation":"dataset","target_id":"reported-dataset-cd51026cdb6a7a"}],"attributes":{"origin":"independent_paper","protocol":"Average across twenty medium- and high-complexity simulated communities.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-026","kind":"evaluation","name":"VirSorter: Simulated prophage-contig detection","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Simulated prophage-contig detection"]},"source_ids":["viral-contig-simulation-2021"],"links":[{"relation":"model","target_id":"reported-model-e78e3886df0d3a"},{"relation":"benchmark","target_id":"reported-task-53e3d216eef6db"},{"relation":"dataset","target_id":"reported-dataset-cd51026cdb6a7a"}],"attributes":{"origin":"independent_paper","protocol":"Average across twenty medium- and high-complexity simulated communities.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-027","kind":"evaluation","name":"GenomeOcean: Natural vs artificial microbial genome sequence","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Natural vs artificial microbial genome sequence"]},"source_ids":["genomeocean-2025"],"links":[{"relation":"model","target_id":"reported-model-2df975e60d16d1"},{"relation":"benchmark","target_id":"reported-task-9f9ab0090f6522"},{"relation":"dataset","target_id":"reported-dataset-0c3ac7efe99c37"}],"attributes":{"origin":"author_reported","protocol":"Source reports natural-versus-artificial sequence classification.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-028","kind":"evaluation","name":"DNABERT-2: Natural vs artificial microbial genome sequence","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Natural vs artificial microbial genome sequence"]},"source_ids":["genomeocean-2025"],"links":[{"relation":"model","target_id":"reported-model-7f1165b35f10e2"},{"relation":"benchmark","target_id":"reported-task-9f9ab0090f6522"},{"relation":"dataset","target_id":"reported-dataset-0c3ac7efe99c37"}],"attributes":{"origin":"independent_paper","protocol":"Source reports natural-versus-artificial sequence classification.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-029","kind":"evaluation","name":"kMetaShot: Mock-community MAG taxonomy classification","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Mock-community MAG taxonomy classification"]},"source_ids":["kmetashot-2025"],"links":[{"relation":"model","target_id":"reported-model-790768ed581685"},{"relation":"benchmark","target_id":"reported-task-8406b6aabfb8c0"},{"relation":"dataset","target_id":"reported-dataset-8cad416ddc80dc"}],"attributes":{"origin":"author_reported","protocol":"Genus classification of MAGs from MegaHIT contigs; uncorrected kMetaShot.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-030","kind":"evaluation","name":"GTDB-Tk: Mock-community MAG taxonomy classification","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Mock-community MAG taxonomy classification"]},"source_ids":["kmetashot-2025"],"links":[{"relation":"model","target_id":"reported-model-3fd1e9f6c573b2"},{"relation":"benchmark","target_id":"reported-task-8406b6aabfb8c0"},{"relation":"dataset","target_id":"reported-dataset-8cad416ddc80dc"}],"attributes":{"origin":"independent_paper","protocol":"Genus classification of the same MAG set.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-031","kind":"evaluation","name":"Lemur: Long-read taxonomic profiling","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Long-read taxonomic profiling"]},"source_ids":["lemur-magnet-2024"],"links":[{"relation":"model","target_id":"reported-model-6ac0730e8481de"},{"relation":"benchmark","target_id":"reported-task-6330d593980b5b"},{"relation":"dataset","target_id":"reported-dataset-768a7ff5bac414"}],"attributes":{"origin":"author_reported","protocol":"Mean across five replicate runs on Zymo LOG 10%.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-032","kind":"evaluation","name":"Kraken 2: Long-read taxonomic profiling","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Long-read taxonomic profiling"]},"source_ids":["lemur-magnet-2024"],"links":[{"relation":"model","target_id":"reported-model-62bc5e5ba13e7d"},{"relation":"benchmark","target_id":"reported-task-6330d593980b5b"},{"relation":"dataset","target_id":"reported-dataset-768a7ff5bac414"}],"attributes":{"origin":"independent_paper","protocol":"Mean across five replicate runs on Zymo LOG 10%.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-033","kind":"evaluation","name":"iPro-MP: Multi-species prokaryotic promoter detection","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Multi-species prokaryotic promoter detection"]},"source_ids":["ipromp-2025"],"links":[{"relation":"model","target_id":"reported-model-9a200c55b0e03e"},{"relation":"benchmark","target_id":"reported-task-a1151e386a3d3f"},{"relation":"dataset","target_id":"reported-dataset-e0f34dcaa1ba3b"}],"attributes":{"origin":"author_reported","protocol":"Average over independent testing sets.","version":null,"comparison":{"protocol_id":null,"dataset_version":"23 test sets","split":"independent test","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-034","kind":"evaluation","name":"Prompt: Multi-species prokaryotic promoter detection","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Multi-species prokaryotic promoter detection"]},"source_ids":["ipromp-2025"],"links":[{"relation":"model","target_id":"reported-model-03080a5289c07e"},{"relation":"benchmark","target_id":"reported-task-a1151e386a3d3f"},{"relation":"dataset","target_id":"reported-dataset-e0f34dcaa1ba3b"}],"attributes":{"origin":"independent_paper","protocol":"Average over the same independent testing sets.","version":null,"comparison":{"protocol_id":null,"dataset_version":"23 test sets","split":"independent test","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-035","kind":"evaluation","name":"ICCTax: Hierarchical metagenomic taxonomy classification","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Hierarchical metagenomic taxonomy classification"]},"source_ids":["icctax-2025"],"links":[{"relation":"model","target_id":"reported-model-67eaf766fa9877"},{"relation":"benchmark","target_id":"reported-task-f4b1c9373f0929"},{"relation":"dataset","target_id":"reported-dataset-561834dfa1682c"}],"attributes":{"origin":"author_reported","protocol":"Macro average precision at genus rank on Complete dataset.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-036","kind":"evaluation","name":"Kraken2: Hierarchical metagenomic taxonomy classification","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Hierarchical metagenomic taxonomy classification"]},"source_ids":["icctax-2025"],"links":[{"relation":"model","target_id":"reported-model-8861b9ad9b9c9b"},{"relation":"benchmark","target_id":"reported-task-f4b1c9373f0929"},{"relation":"dataset","target_id":"reported-dataset-561834dfa1682c"}],"attributes":{"origin":"independent_paper","protocol":"Macro average precision at genus rank on Complete dataset.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-037","kind":"evaluation","name":"Chai-1: Antibody–antigen interaction prediction using folded complexes","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Antibody–antigen interaction prediction using folded complexes"]},"source_ids":["antibody-flexibility-2025"],"links":[{"relation":"model","target_id":"reported-model-b3fdf259d51533"},{"relation":"benchmark","target_id":"reported-task-0c92cda11228c4"},{"relation":"dataset","target_id":"reported-dataset-ee26acbd6e8cf7"}],"attributes":{"origin":"independent_paper","protocol":"Interaction classifier evaluated using Chai-1-folded input complexes; this is pipeline AUC, not DockQ.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-038","kind":"evaluation","name":"Boltz-1: Antibody–antigen interaction prediction using folded complexes","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Antibody–antigen interaction prediction using folded complexes"]},"source_ids":["antibody-flexibility-2025"],"links":[{"relation":"model","target_id":"reported-model-884582fb0c70dc"},{"relation":"benchmark","target_id":"reported-task-0c92cda11228c4"},{"relation":"dataset","target_id":"reported-dataset-ee26acbd6e8cf7"}],"attributes":{"origin":"independent_paper","protocol":"Interaction classifier evaluated using Boltz-1-folded input complexes; this is pipeline AUC, not DockQ.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-039","kind":"evaluation","name":"Boltz-1: Protein–ligand pose prediction","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand pose prediction"]},"source_ids":["boltz1-2025"],"links":[{"relation":"model","target_id":"reported-model-4058c43eb73b90"},{"relation":"benchmark","target_id":"reported-task-c04bb5ee6ecea6"},{"relation":"dataset","target_id":"reported-dataset-e45a5a140888ee"}],"attributes":{"origin":"author_reported","protocol":"Highest-confidence pose from five samples; precomputed MSAs up to 4,096 sequences.","version":"3 recycling rounds; 200 diffusion steps","comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-040","kind":"evaluation","name":"Ibex: Antibody loop structure prediction","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Antibody loop structure prediction"]},"source_ids":["ibex-2025"],"links":[{"relation":"model","target_id":"reported-model-2ae5fb0c147618"},{"relation":"benchmark","target_id":"reported-task-f3a12dbc0e0439"},{"relation":"dataset","target_id":"reported-dataset-b7204b005bd476"}],"attributes":{"origin":"author_reported","protocol":"Backbone RMSD after framework alignment; average over antibody test structures.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-041","kind":"evaluation","name":"Chai-1: Antibody loop structure prediction","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Antibody loop structure prediction"]},"source_ids":["ibex-2025"],"links":[{"relation":"model","target_id":"reported-model-70c732770a200f"},{"relation":"benchmark","target_id":"reported-task-f3a12dbc0e0439"},{"relation":"dataset","target_id":"reported-dataset-b7204b005bd476"}],"attributes":{"origin":"independent_paper","protocol":"Backbone RMSD after framework alignment; one seed and one diffusion trajectory.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-042","kind":"evaluation","name":"DEELIG: Protein–ligand binding affinity prediction","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand binding affinity prediction"]},"source_ids":["deelig-2021"],"links":[{"relation":"model","target_id":"reported-model-fa2da404b4d08e"},{"relation":"benchmark","target_id":"reported-task-d81be76396e644"},{"relation":"dataset","target_id":"reported-dataset-327cfcdae0c937"}],"attributes":{"origin":"author_reported","protocol":"Source paper reports DEELIG on PDBbind core set.","version":null,"comparison":{"protocol_id":null,"dataset_version":"v2016","split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-043","kind":"evaluation","name":"TOPBP (Complex): Protein–ligand binding affinity prediction","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand binding affinity prediction"]},"source_ids":["deelig-2021"],"links":[{"relation":"model","target_id":"reported-model-0068c3eff1bf7b"},{"relation":"benchmark","target_id":"reported-task-d81be76396e644"},{"relation":"dataset","target_id":"reported-dataset-327cfcdae0c937"}],"attributes":{"origin":"paper_compilation","protocol":"Source table compiles a previously published comparator; protocol equivalence is not established.","version":null,"comparison":{"protocol_id":null,"dataset_version":"v2016","split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","original_evaluation":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-044","kind":"evaluation","name":"MolAS: Physically valid protein–ligand pose selection","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Physically valid protein–ligand pose selection"]},"source_ids":["molas-2026"],"links":[{"relation":"model","target_id":"reported-model-0eb4b0535b58e3"},{"relation":"benchmark","target_id":"reported-task-d1c46526c39983"},{"relation":"dataset","target_id":"reported-dataset-4b6c13924d4256"}],"attributes":{"origin":"author_reported","protocol":"Averaged five-fold algorithm-selection performance on PoseBusters; joint RMSD and validity criterion.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-045","kind":"evaluation","name":"Single best solver: Physically valid protein–ligand pose selection","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Physically valid protein–ligand pose selection"]},"source_ids":["molas-2026"],"links":[{"relation":"model","target_id":"reported-model-028e4bb9baa074"},{"relation":"benchmark","target_id":"reported-task-d1c46526c39983"},{"relation":"dataset","target_id":"reported-dataset-4b6c13924d4256"}],"attributes":{"origin":"independent_paper","protocol":"Single best solver baseline under the same averaged five-fold selection test.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-046","kind":"evaluation","name":"AutoDock Vina holo: Intrinsically disordered protein ensemble docking","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Intrinsically disordered protein ensemble docking"]},"source_ids":["ensemble-idp-docking-2025"],"links":[{"relation":"model","target_id":"reported-model-1587ab674d30a2"},{"relation":"benchmark","target_id":"reported-task-dec9e0f5e3da2a"},{"relation":"dataset","target_id":"reported-dataset-00201f65c32f6d"}],"attributes":{"origin":"independent_paper","protocol":"Fraction of docked frames within 3 Å of MD-observed bound pose; holo protein ensemble.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-047","kind":"evaluation","name":"DiffDock holo: Intrinsically disordered protein ensemble docking","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Intrinsically disordered protein ensemble docking"]},"source_ids":["ensemble-idp-docking-2025"],"links":[{"relation":"model","target_id":"reported-model-75e4e5e5965320"},{"relation":"benchmark","target_id":"reported-task-dec9e0f5e3da2a"},{"relation":"dataset","target_id":"reported-dataset-00201f65c32f6d"}],"attributes":{"origin":"independent_paper","protocol":"Fraction of docked frames within 3 Å of MD-observed bound pose; holo protein ensemble.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-048","kind":"evaluation","name":"AK-score-ensemble: Protein–ligand binding affinity scoring","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand binding affinity scoring"]},"source_ids":["akscore-2020"],"links":[{"relation":"model","target_id":"reported-model-54be8a811c206e"},{"relation":"benchmark","target_id":"reported-task-a78312d5df6dad"},{"relation":"dataset","target_id":"reported-dataset-f18fcc23dfa798"}],"attributes":{"origin":"author_reported","protocol":"CASF-2016 scoring-power evaluation.","version":"ensemble; learning rate 0.0007","comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-049","kind":"evaluation","name":"AK-score-single: Protein–ligand binding affinity scoring","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand binding affinity scoring"]},"source_ids":["akscore-2020"],"links":[{"relation":"model","target_id":"reported-model-de89576d8d316b"},{"relation":"benchmark","target_id":"reported-task-a78312d5df6dad"},{"relation":"dataset","target_id":"reported-dataset-f18fcc23dfa798"}],"attributes":{"origin":"author_reported","protocol":"CASF-2016 scoring-power evaluation.","version":"single; learning rate 0.0007","comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-050","kind":"evaluation","name":"PMF + ECFP + PF (LightGBM): Protein–ligand binding energy prediction","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand binding energy prediction"]},"source_ids":["fingerprint-scoring-2022"],"links":[{"relation":"model","target_id":"reported-model-3c196326586fa7"},{"relation":"benchmark","target_id":"reported-task-94802534b7026d"},{"relation":"dataset","target_id":"reported-dataset-16d01b5ef88e84"}],"attributes":{"origin":"author_reported","protocol":"Binding-energy model using ligand and protein fingerprints with LightGBM.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-051","kind":"evaluation","name":"PMF (LASSO): Protein–ligand binding energy prediction","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand binding energy prediction"]},"source_ids":["fingerprint-scoring-2022"],"links":[{"relation":"model","target_id":"reported-model-8100b3de6c7811"},{"relation":"benchmark","target_id":"reported-task-94802534b7026d"},{"relation":"dataset","target_id":"reported-dataset-16d01b5ef88e84"}],"attributes":{"origin":"author_reported","protocol":"PMF-only LASSO baseline evaluated by the same authors.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-001","kind":"evaluation","name":"ARSENAL+ChromBPNet: regulatory-variant scoring","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["regulatory-variant scoring"]},"source_ids":["arsenal-regulatory-dna-2026"],"links":[{"relation":"model","target_id":"reported-model-ee1ae8162c7d67"},{"relation":"benchmark","target_id":"reported-task-b9199a30a0bcb2"},{"relation":"dataset","target_id":"reported-dataset-158b121281b650"}],"attributes":{"origin":"author_reported","protocol":"Supervised ChromBPNet variant scoring with ARSENAL motif-discovery regularization","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-002","kind":"evaluation","name":"PlantCAD2: cross-species conservation prediction","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["cross-species conservation prediction"]},"source_ids":["plantcad2-2025"],"links":[{"relation":"model","target_id":"reported-model-65059c3a806306"},{"relation":"benchmark","target_id":"reported-task-3109f8d0f2b7b5"},{"relation":"dataset","target_id":"reported-dataset-b6d0ebaca196a6"}],"attributes":{"origin":"author_reported","protocol":"Zero-shot score for conserved versus non-conserved sites from alignments of 35 Andropogoneae genomes","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-003","kind":"evaluation","name":"Stacking-Auto: enhancer prediction","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["enhancer prediction"]},"source_ids":["hi-enhancer-2025"],"links":[{"relation":"model","target_id":"reported-model-0829aff5471d4b"},{"relation":"benchmark","target_id":"reported-task-22024610c4d658"},{"relation":"dataset","target_id":"reported-dataset-a8610f2b80cdf0"}],"attributes":{"origin":"author_reported","protocol":"Two-stage Hi-Enhancer system; paper Table 2 method comparison","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-004","kind":"evaluation","name":"position-aware CNN: enhancer prediction","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["enhancer prediction"]},"source_ids":["enhancer-position-encoding-2024"],"links":[{"relation":"model","target_id":"reported-model-d2c81acf1c42c4"},{"relation":"benchmark","target_id":"reported-task-64607443a9ba15"},{"relation":"dataset","target_id":"reported-dataset-a03b8e9efde37b"}],"attributes":{"origin":"author_reported","protocol":"Nucleotide position-aware feature encoding; average assessment of CNN classifier","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-005","kind":"evaluation","name":"ADAR-GPT continual: A-to-I RNA editing site prediction","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["A-to-I RNA editing site prediction"]},"source_ids":["adar-gpt-editing-2026"],"links":[{"relation":"model","target_id":"reported-model-1d2aa9880a1c77"},{"relation":"benchmark","target_id":"reported-task-d635fc6c281a27"},{"relation":"dataset","target_id":"reported-dataset-236eaa4e55147f"}],"attributes":{"origin":"author_reported","protocol":"Curriculum plus 15% fine-tuning; 201-nt sequence windows; decision threshold 0.5","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":"15% validation set","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-006","kind":"evaluation","name":"R3Design: RNA sequence design","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA sequence design"]},"source_ids":["r3design-2025"],"links":[{"relation":"model","target_id":"reported-model-1b5fa066945d3d"},{"relation":"benchmark","target_id":"reported-task-df18c710f45213"},{"relation":"dataset","target_id":"reported-dataset-71614d99b3099f"}],"attributes":{"origin":"author_reported","protocol":"Tertiary-structure-conditioned RNA sequence design; external Rfam assessment","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":"external","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-007","kind":"evaluation","name":"CUPID Data-aug-Avg: non-coding RNA pairwise interaction prediction","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["non-coding RNA pairwise interaction prediction"]},"source_ids":["cupid-rna-interactions-2026"],"links":[{"relation":"model","target_id":"reported-model-2894d253c5e8a8"},{"relation":"benchmark","target_id":"reported-task-c7ce06b753b8b6"},{"relation":"dataset","target_id":"reported-dataset-32ccef507a1dd7"}],"attributes":{"origin":"author_reported","protocol":"Data augmentation with average pooling for molecule-level ncRNA embeddings","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-008","kind":"evaluation","name":"ProteinBERT LLM-encoding model: mRNA-protein interaction prediction","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["mRNA-protein interaction prediction"]},"source_ids":["mrna-protein-diversity-2026"],"links":[{"relation":"model","target_id":"reported-model-41ae49bb40ed8e"},{"relation":"benchmark","target_id":"reported-task-d7e6274011946e"},{"relation":"dataset","target_id":"reported-dataset-2e87449871ca47"}],"attributes":{"origin":"author_reported","protocol":"LLM encoding of protein partner; RBP-aware partition tests generalization to unseen protein diversity","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":"RBP-aware test set","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-009","kind":"evaluation","name":"ESM2 650M: human-versus-viral protein classification","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["human-versus-viral protein classification"]},"source_ids":["viral-immune-mimicry-2025"],"links":[{"relation":"model","target_id":"reported-model-4c73500c39e9d0"},{"relation":"benchmark","target_id":"reported-task-53506fe386e4a1"},{"relation":"dataset","target_id":"reported-dataset-43f24c4dfb7351"}],"attributes":{"origin":"author_reported","protocol":"ESM2 650M embedding-based human-virus classifier","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-010","kind":"evaluation","name":"ProtT5 embeddings + ensemble classifier: protein-protein binding-site prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["protein-protein binding-site prediction"]},"source_ids":["protein-binding-sites-2023"],"links":[{"relation":"model","target_id":"reported-model-fdac4c1ec8a433"},{"relation":"benchmark","target_id":"reported-task-f0ed5188dbb6d4"},{"relation":"dataset","target_id":"reported-dataset-f08b1a60aebeeb"}],"attributes":{"origin":"author_reported","protocol":"Explainable ensemble binding-site predictor using ProtT5 features","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-011","kind":"evaluation","name":"CLAPE-SMB with ESM-2: protein-small molecule binding-site prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["protein-small molecule binding-site prediction"]},"source_ids":["clape-smb-2024"],"links":[{"relation":"model","target_id":"reported-model-57dbab30462150"},{"relation":"benchmark","target_id":"reported-task-b181ed450cdd41"},{"relation":"dataset","target_id":"reported-dataset-701d910b02d25c"}],"attributes":{"origin":"author_reported","protocol":"Contrastive CLAPE-SMB binding-site predictor with ESM-2 feature extractor","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-012","kind":"evaluation","name":"Vaxign-DL + ESM: vaccine-antigen candidate prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["vaccine-antigen candidate prediction"]},"source_ids":["vaxign-esm-2024"],"links":[{"relation":"model","target_id":"reported-model-8ad3e0cefde796"},{"relation":"benchmark","target_id":"reported-task-47465954d606e6"},{"relation":"dataset","target_id":"reported-dataset-b91c871eb7740a"}],"attributes":{"origin":"author_reported","protocol":"Combined skip architecture, four layers, ESM-generated sequence features","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-013","kind":"evaluation","name":"scGPT + residual geometry: gene-regulatory signal prediction","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["gene-regulatory signal prediction"]},"source_ids":["single-cell-residual-geometry-2026"],"links":[{"relation":"model","target_id":"reported-model-d6a7fa854437e8"},{"relation":"benchmark","target_id":"reported-task-99afd88cb12895"},{"relation":"dataset","target_id":"reported-dataset-d9fdd8dc7a0184"}],"attributes":{"origin":"author_reported","protocol":"Asymmetric extraction, PCA-64 centered cosine geometry added to scGPT baseline","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-014","kind":"evaluation","name":"GREmLN: cell-type annotation","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["cell-type annotation"]},"source_ids":["gremln-2026"],"links":[{"relation":"model","target_id":"reported-model-53d6515bcc1f39"},{"relation":"benchmark","target_id":"reported-task-6312c8a7ac045e"},{"relation":"dataset","target_id":"reported-dataset-59def895fbdbb4"}],"attributes":{"origin":"author_reported","protocol":"Zero-shot cell-type annotation using pre-trained cellular graph foundation model","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":"zero-shot","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-015","kind":"evaluation","name":"Cell-DINO ViT-L: protein localization classification","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["protein localization classification"]},"source_ids":["cell-dino-2025"],"links":[{"relation":"model","target_id":"reported-model-808b23c65fbc89"},{"relation":"benchmark","target_id":"reported-task-7621fa1be55362"},{"relation":"dataset","target_id":"reported-dataset-d356eac961cb69"}],"attributes":{"origin":"author_reported","protocol":"Self-supervised microscopy embedding pre-trained on HPA-FoV; downstream protein-localization classifier. Dataset-specific pretraining; the paper does not claim a general-purpose foundation model that generalizes beyond these benchmarks.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-016","kind":"evaluation","name":"scGen: differentially expressed gene identification","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["differentially expressed gene identification"]},"source_ids":["insilico-perturbation-auprc-2025"],"links":[{"relation":"model","target_id":"reported-model-fcf2cd29a81aae"},{"relation":"benchmark","target_id":"reported-task-003d746a129c9b"},{"relation":"dataset","target_id":"reported-dataset-7bf2cf7d2b2d01"}],"attributes":{"origin":"independent_paper","protocol":"In-silico perturbation assessment with precision sampled at fixed 50% recall","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":"CD14+Mono","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-017","kind":"evaluation","name":"TCINet + HTRS: pathogen detection","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["pathogen detection"]},"source_ids":["metagenomic-pathogens-2025"],"links":[{"relation":"model","target_id":"reported-model-4ce8cae0f2eafc"},{"relation":"benchmark","target_id":"reported-task-d3fd502fdc2b38"},{"relation":"dataset","target_id":"reported-dataset-1c4c71078ffe01"}],"attributes":{"origin":"author_reported","protocol":"Taxonomy-constrained inference network with hierarchical taxonomy representation","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-018","kind":"evaluation","name":"DETIRE: viral sequence detection","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["viral sequence detection"]},"source_ids":["detire-viral-metagenomes-2023"],"links":[{"relation":"model","target_id":"reported-model-6d9dbac97852d8"},{"relation":"benchmark","target_id":"reported-task-3d4dec23120fef"},{"relation":"dataset","target_id":"reported-dataset-0b54f42a987b1d"}],"attributes":{"origin":"author_reported","protocol":"Hybrid deep learning virus-fragment classifier on paper testing dataset","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":"test","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-019","kind":"evaluation","name":"PC-mer + LR: metagenomic genus classification","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["metagenomic genus classification"]},"source_ids":["pc-mer-2024"],"links":[{"relation":"model","target_id":"reported-model-688eb780ef7d2e"},{"relation":"benchmark","target_id":"reported-task-4420dcdfe8338d"},{"relation":"dataset","target_id":"reported-dataset-d28955d5872903"}],"attributes":{"origin":"author_reported","protocol":"k=8 PC-mer feature extraction with logistic regression on AMP genus-classification dataset","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":"genus-level","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-020","kind":"evaluation","name":"MDL4Microbiome: microbiome disease-state classification","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["microbiome disease-state classification"]},"source_ids":["mdl4microbiome-2022"],"links":[{"relation":"model","target_id":"reported-model-e6ba198c2ac996"},{"relation":"benchmark","target_id":"reported-task-e2009c35eabd69"},{"relation":"dataset","target_id":"reported-dataset-bd3f98e2eeb5d3"}],"attributes":{"origin":"author_reported","protocol":"Multimodal deep learning model on colorectal-cancer versus healthy microbiome samples","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-021","kind":"evaluation","name":"binding-affinity meta-model: protein-ligand binding affinity prediction","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["protein-ligand binding affinity prediction"]},"source_ids":["ligand-affinity-meta-model-2024"],"links":[{"relation":"model","target_id":"reported-model-df0efcc0346224"},{"relation":"benchmark","target_id":"reported-task-77a32496ce8fe6"},{"relation":"dataset","target_id":"reported-dataset-17132fbabd7683"}],"attributes":{"origin":"author_reported","protocol":"Sequence-or-structure meta-model; predicts ln(Kd/Ki) using docked and deep-learning components","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":"core benchmark","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-022","kind":"evaluation","name":"DeepInterAware: antigen-antibody HIV neutralization prediction","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["antigen-antibody HIV neutralization prediction"]},"source_ids":["deepinteraware-2025"],"links":[{"relation":"model","target_id":"reported-model-8d2c291733dfe1"},{"relation":"benchmark","target_id":"reported-task-45ead9a1eddf8d"},{"relation":"dataset","target_id":"reported-dataset-50f0bdb7cf9ca4"}],"attributes":{"origin":"author_reported","protocol":"Sequence-based interface-aware model, antibody-unseen split","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":"antibody-unseen","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-023","kind":"evaluation","name":"TransBind: transcription-factor DNA binding-site prediction","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["transcription-factor DNA binding-site prediction"]},"source_ids":["transbind-2026"],"links":[{"relation":"model","target_id":"reported-model-81b0394d5ac3e8"},{"relation":"benchmark","target_id":"reported-task-ac191e878dff5e"},{"relation":"dataset","target_id":"reported-dataset-034c60a2dabc73"}],"attributes":{"origin":"author_reported","protocol":"Integrates protein and DNA embeddings for TFBS prediction on paper test dataset","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":"test","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-024","kind":"evaluation","name":"ESM2_AMPS: protein-protein interaction prediction","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["protein-protein interaction prediction"]},"source_ids":["esm2-amp-2025"],"links":[{"relation":"model","target_id":"reported-model-67ea6bd77b2ed1"},{"relation":"benchmark","target_id":"reported-task-09c3100b77dcc5"},{"relation":"dataset","target_id":"reported-dataset-9135087a16af1c"}],"attributes":{"origin":"author_reported","protocol":"ESM2-derived embeddings plus paper interaction predictor","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evidence-alphafold-docs-input","kind":"source","name":"AlphaFold 3 input specification","description":"Primary source inspected for the AlphaFold 3 profile; source checking is not experimental reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/google-deepmind/alphafold3/blob/c0f97eda2f1f482fd94d3a38bece18c7069b4a5c/docs/input.md","version":"c0f97eda2f1f482fd94d3a38bece18c7069b4a5c","retrieved_at":"2026-09-16T19:56:24.856478+00:00","artifact_url":"https://raw.githubusercontent.com/google-deepmind/alphafold3/c0f97eda2f1f482fd94d3a38bece18c7069b4a5c/docs/input.md","artifact_sha256":"e75407473d8c6975a91bd40605b7c8075008426a15a1a952826996e3387226c1","source_locator":"docs/input.md","extraction_method":"primary_artifact_review"}} {"id":"evidence-alphafold-docs-installation","kind":"source","name":"AlphaFold 3 installation documentation","description":"Primary source inspected for the AlphaFold 3 profile; source checking is not experimental reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/google-deepmind/alphafold3/blob/c0f97eda2f1f482fd94d3a38bece18c7069b4a5c/docs/installation.md","version":"c0f97eda2f1f482fd94d3a38bece18c7069b4a5c","retrieved_at":"2026-09-16T19:56:24.856478+00:00","artifact_url":"https://raw.githubusercontent.com/google-deepmind/alphafold3/c0f97eda2f1f482fd94d3a38bece18c7069b4a5c/docs/installation.md","artifact_sha256":"4e91ac50393d9579316ca13cf438d516a17cea2f73c2ef1a0a1066b37c797c34","source_locator":"docs/installation.md","extraction_method":"primary_artifact_review"}} {"id":"evidence-alphafold-docs-output","kind":"source","name":"AlphaFold 3 output specification","description":"Primary source inspected for the AlphaFold 3 profile; source checking is not experimental reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/google-deepmind/alphafold3/blob/c0f97eda2f1f482fd94d3a38bece18c7069b4a5c/docs/output.md","version":"c0f97eda2f1f482fd94d3a38bece18c7069b4a5c","retrieved_at":"2026-09-16T19:56:24.856478+00:00","artifact_url":"https://raw.githubusercontent.com/google-deepmind/alphafold3/c0f97eda2f1f482fd94d3a38bece18c7069b4a5c/docs/output.md","artifact_sha256":"bb792ca2564e5a1f48512388efc4ffda00535f8847211e095f11ce15111362fe","source_locator":"docs/output.md","extraction_method":"primary_artifact_review"}} {"id":"evidence-alphafold-docs-performance","kind":"source","name":"AlphaFold 3 performance documentation","description":"Primary source inspected for the AlphaFold 3 profile; source checking is not experimental reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/google-deepmind/alphafold3/blob/c0f97eda2f1f482fd94d3a38bece18c7069b4a5c/docs/performance.md","version":"c0f97eda2f1f482fd94d3a38bece18c7069b4a5c","retrieved_at":"2026-09-16T19:56:24.856478+00:00","artifact_url":"https://raw.githubusercontent.com/google-deepmind/alphafold3/c0f97eda2f1f482fd94d3a38bece18c7069b4a5c/docs/performance.md","artifact_sha256":"5fb82c4be3c91f6d488196a57be03dab6a3685d629c87c72d719bc868fc9e38a","source_locator":"docs/performance.md","extraction_method":"primary_artifact_review"}} {"id":"evidence-alphafold-license","kind":"source","name":"AlphaFold 3 code licence","description":"Primary source inspected for the AlphaFold 3 profile; source checking is not experimental reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/google-deepmind/alphafold3/blob/c0f97eda2f1f482fd94d3a38bece18c7069b4a5c/LICENSE","version":"c0f97eda2f1f482fd94d3a38bece18c7069b4a5c","retrieved_at":"2026-09-16T19:56:24.856478+00:00","artifact_url":"https://raw.githubusercontent.com/google-deepmind/alphafold3/c0f97eda2f1f482fd94d3a38bece18c7069b4a5c/LICENSE","artifact_sha256":"cfc7749b96f63bd31c3c42b5c471bf756814053e847c10f3eb003417bc523d30","source_locator":"LICENSE","extraction_method":"primary_artifact_review"}} {"id":"evidence-alphafold-paper","kind":"source","name":"AlphaFold 3 paper","description":"Primary source inspected for the AlphaFold 3 profile; source checking is not experimental reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11168924/","version":"Nature 2024; DOI 10.1038/s41586-024-07487-w; XML retrieved 2026-09-16","retrieved_at":"2026-09-16T19:56:24.856478+00:00","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11168924/fullTextXML","artifact_sha256":"e627083f74d990b275e96a7503eee3e3eae9e3035b8efb28321277fcac6f1d6e","source_locator":"Main text, Fig. 1; Model limitations; Methods","extraction_method":"primary_artifact_review"}} {"id":"evidence-alphafold-readme","kind":"source","name":"AlphaFold 3 README","description":"Primary source inspected for the AlphaFold 3 profile; source checking is not experimental reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/google-deepmind/alphafold3/blob/c0f97eda2f1f482fd94d3a38bece18c7069b4a5c/README.md","version":"c0f97eda2f1f482fd94d3a38bece18c7069b4a5c","retrieved_at":"2026-09-16T19:56:24.856478+00:00","artifact_url":"https://raw.githubusercontent.com/google-deepmind/alphafold3/c0f97eda2f1f482fd94d3a38bece18c7069b4a5c/README.md","artifact_sha256":"cc16ba436ea8a967c764ac034a655d632b4c1ef8b4065868b8d07914c7cb88be","source_locator":"README.md","extraction_method":"primary_artifact_review"}} {"id":"evidence-alphafold-server-faq","kind":"source","name":"AlphaFold Server FAQ","description":"Primary source inspected for the AlphaFold 3 profile; source checking is not experimental reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://alphafoldserver.com/faq","version":"Rendered public FAQ, 2026-09-16","retrieved_at":"2026-09-16T19:56:24.856478+00:00","artifact_url":"https://alphafoldserver.com/faq","artifact_sha256":"66df2a77d57bbf01afc72d6ac110bc195d85a259d01d3f532c2e6e576e652eb0","source_locator":"Named FAQ questions; browser-rendered body text","extraction_method":"rendered_primary_page_review"}} {"id":"evidence-alphafold-server-output-terms","kind":"source","name":"AlphaFold Server output terms","description":"Primary source inspected for the AlphaFold 3 profile; source checking is not experimental reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://alphafoldserver.com/output-terms","version":"Last modified 2024-05-08; rendered 2026-09-16","retrieved_at":"2026-09-16T19:56:24.856478+00:00","artifact_url":"https://alphafoldserver.com/output-terms","artifact_sha256":"9832af4e56499e0f373584b95327dfc805b834e90186f85ee2387f6e1509c126","source_locator":"Use restrictions; Miscellaneous","extraction_method":"rendered_primary_page_review"}} {"id":"evidence-alphafold-server-terms","kind":"source","name":"AlphaFold Server terms","description":"Primary source inspected for the AlphaFold 3 profile; source checking is not experimental reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://alphafoldserver.com/terms","version":"Last modified 2024-05-08; rendered 2026-09-16","retrieved_at":"2026-09-16T19:56:24.856478+00:00","artifact_url":"https://alphafoldserver.com/terms","artifact_sha256":"5861ed425310fcda7e2a7807d20fbfd4e6bf6dc00801d082cd8b4f2507dacfae","source_locator":"Key things to know; Overview","extraction_method":"rendered_primary_page_review"}} {"id":"evidence-alphafold-supplement","kind":"source","name":"AlphaFold 3 supplementary information","description":"Original supplementary PDF inspected for architecture and training-data provenance; table layout visually checked.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://www.nature.com/articles/s41586-024-07487-w#Sec23","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11168924/supplementaryFiles","archive_member":"41586_2024_7487_MOESM1_ESM.pdf","artifact_sha256":"9f19a51ea050e6f77b6c07e2c06aa871d40050fd07da22ae09e92d0887a83130","archive_sha256":"b175fcf15fecf03fbf2ec39e47bd59a504af601080b35c32f3cdee3507ff4e88","retrieved_at":"2026-09-16T20:31:11.456977+00:00","version":"Supplement distributed with DOI 10.1038/s41586-024-07487-w, retrieved 2026-09-16","locator":"Sections 2.2, 2.5, 3 and 5.2; Tables 3 and 6","review_method":"automated_source_review","hash_scope":"SHA-256 of original PDF bytes extracted from the Europe PMC supplement archive"}} {"id":"evidence-alphafold-weights-terms-of-use","kind":"source","name":"AlphaFold 3 weights terms","description":"Primary source inspected for the AlphaFold 3 profile; source checking is not experimental reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/google-deepmind/alphafold3/blob/c0f97eda2f1f482fd94d3a38bece18c7069b4a5c/WEIGHTS_TERMS_OF_USE.md","version":"c0f97eda2f1f482fd94d3a38bece18c7069b4a5c","retrieved_at":"2026-09-16T19:56:24.856478+00:00","artifact_url":"https://raw.githubusercontent.com/google-deepmind/alphafold3/c0f97eda2f1f482fd94d3a38bece18c7069b4a5c/WEIGHTS_TERMS_OF_USE.md","artifact_sha256":"41adf62ff5eabc58831c828793988537948663c139f8b87d8d413851b150b6e5","source_locator":"WEIGHTS_TERMS_OF_USE.md","extraction_method":"primary_artifact_review"}} {"id":"evidence-benchmark-cafa-20260916","kind":"source","name":"CAFA official description — reviewed snapshot 2026-09-16","description":"Official CAFA page snapshot inspected in this completion pass; the cached bytes differ from the prior source snapshot.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"d7a5fa76bea98551f8322aab9da965c383104c79d83992e723c0c5836b2fdcd1","licence":null,"locator":"HTML main page","missing_metadata":{"immutable_source_version":"unavailable","licence":"unextracted","version":"unreported"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T19:47:27.547276+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","scope_note":"Mutable official page: content hash and retrieval time recorded, but the original page is not redistributed.","url":"https://biofunctionprediction.org/cafa/","version":"2026-09-16 website snapshot sha256:d7a5fa76bea98551f8322aab9da965c383104c79d83992e723c0c5836b2fdcd1","artifact_url":"https://biofunctionprediction.org/cafa/"}} {"id":"evidence-benchmark-cami-snapshot","kind":"source","name":"cami official source","description":"Primary project documentation or project-maintained evidence.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"17825bf33280f40b596a104c547b57fae5ee5c07f8d60b396d0f4780d47ef9a5","licence":null,"locator":"HTML main page","missing_metadata":{"immutable_source_version":"unavailable","licence":"unextracted","version":"unreported"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:31:54.339676+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","scope_note":"Mutable official page: content hash and retrieval time recorded, but the original page is not redistributed.","url":"https://cami-challenge.org/","version":"Retrieved website snapshot sha256:17825bf33280f40b596a104c547b57fae5ee5c07f8d60b396d0f4780d47ef9a5","artifact_url":"https://cami-challenge.org/"}} {"id":"evidence-benchmark-capri-snapshot","kind":"source","name":"capri official source","description":"Primary project documentation or project-maintained evidence.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"5e22606895cf0c30565ed4456bc680ca97c10c8c5d38054f82a6e4677c8d8c24","licence":null,"locator":"HTML main page","missing_metadata":{"immutable_source_version":"unavailable","licence":"unextracted","version":"unreported"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:31:54.812767+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","scope_note":"Mutable official page: content hash and retrieval time recorded, but the original page is not redistributed.","url":"https://www.capri-docking.org/","version":"Retrieved website snapshot sha256:5e22606895cf0c30565ed4456bc680ca97c10c8c5d38054f82a6e4677c8d8c24","artifact_url":"https://www.capri-docking.org/"}} {"id":"evidence-benchmark-casp-snapshot","kind":"source","name":"casp official source","description":"Primary project documentation or project-maintained evidence.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"de391b68636462ddb78d0659d8784128909d23e9c015832cca94d883f404a3f7","licence":null,"locator":"HTML main page","missing_metadata":{"immutable_source_version":"unavailable","licence":"unextracted","version":"unreported"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:31:54.981498+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","scope_note":"Mutable official page: content hash and retrieval time recorded, but the original page is not redistributed.","url":"https://predictioncenter.org/","version":"Retrieved website snapshot sha256:de391b68636462ddb78d0659d8784128909d23e9c015832cca94d883f404a3f7","artifact_url":"https://predictioncenter.org/"}} {"id":"evidence-benchmark-flip2-snapshot","kind":"source","name":"flip2 official source","description":"Primary project documentation or project-maintained evidence.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"cd991ae6e76a5e84ea5449f91c4ed86ba4f57942682dc3c50d872c366bbbd4b7","licence":null,"locator":"HTML main page","missing_metadata":{"immutable_source_version":"unavailable","licence":"unextracted","version":"unreported"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:31:54.381289+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","scope_note":"Mutable official page: content hash and retrieval time recorded, but the original page is not redistributed.","url":"https://flip.protein.properties/","version":"Retrieved website snapshot sha256:cd991ae6e76a5e84ea5449f91c4ed86ba4f57942682dc3c50d872c366bbbd4b7","artifact_url":"https://flip.protein.properties/"}} {"id":"evidence-benchmark-glycan-classification-dataset","kind":"source","name":"GlycanML/GlycanML module/custom_datasets/glycan_classification.py","description":"Pinned computational evaluator or dataset/configuration source inspected for benchmark metadata.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://raw.githubusercontent.com/GlycanML/GlycanML/9f392aa6f9c6d74a296a250199beb347923d04e0/module/custom_datasets/glycan_classification.py","artifact_url":"https://raw.githubusercontent.com/GlycanML/GlycanML/9f392aa6f9c6d74a296a250199beb347923d04e0/module/custom_datasets/glycan_classification.py","artifact_sha256":"1a385c4afcbfc3d2f1f319cf03fc0f4be5a1e9848cd51f9df4239cbd39771530","version":"9f392aa6f9c6d74a296a250199beb347923d04e0","retrieved_at":"2026-09-16T20:43:24.344407+00:00"}} {"id":"evidence-benchmark-glycan-immunogenicity-dataset","kind":"source","name":"GlycanML/GlycanML module/custom_datasets/glycan_immunogenicity.py","description":"Pinned computational evaluator or dataset/configuration source inspected for benchmark metadata.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://raw.githubusercontent.com/GlycanML/GlycanML/9f392aa6f9c6d74a296a250199beb347923d04e0/module/custom_datasets/glycan_immunogenicity.py","artifact_url":"https://raw.githubusercontent.com/GlycanML/GlycanML/9f392aa6f9c6d74a296a250199beb347923d04e0/module/custom_datasets/glycan_immunogenicity.py","artifact_sha256":"73bb768deff2b22b4cf57199353267dc0778b42ccbbaee8035e91ec67b3a3a1f","version":"9f392aa6f9c6d74a296a250199beb347923d04e0","retrieved_at":"2026-09-16T20:43:24.817696+00:00"}} {"id":"evidence-benchmark-glycan-interaction-dataset","kind":"source","name":"GlycanML/GlycanML module/custom_datasets/glycan_interaction.py","description":"Pinned computational evaluator or dataset/configuration source inspected for benchmark metadata.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://raw.githubusercontent.com/GlycanML/GlycanML/9f392aa6f9c6d74a296a250199beb347923d04e0/module/custom_datasets/glycan_interaction.py","artifact_url":"https://raw.githubusercontent.com/GlycanML/GlycanML/9f392aa6f9c6d74a296a250199beb347923d04e0/module/custom_datasets/glycan_interaction.py","artifact_sha256":"e7b1fcae29941ffef3cc6cefc5df373a05fce4d6cf9ffe46cde594503fb2b7f5","version":"9f392aa6f9c6d74a296a250199beb347923d04e0","retrieved_at":"2026-09-16T20:43:24.579180+00:00"}} {"id":"evidence-benchmark-glycan-link-dataset","kind":"source","name":"GlycanML/GlycanML module/custom_datasets/glycan_link.py","description":"Pinned computational evaluator or dataset/configuration source inspected for benchmark metadata.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://raw.githubusercontent.com/GlycanML/GlycanML/9f392aa6f9c6d74a296a250199beb347923d04e0/module/custom_datasets/glycan_link.py","artifact_url":"https://raw.githubusercontent.com/GlycanML/GlycanML/9f392aa6f9c6d74a296a250199beb347923d04e0/module/custom_datasets/glycan_link.py","artifact_sha256":"eab9f95d0007d0c4cc903530aa6677c7d8f73734d7826f6d73ba1b0f5a9319d8","version":"9f392aa6f9c6d74a296a250199beb347923d04e0","retrieved_at":"2026-09-16T20:43:24.095380+00:00"}} {"id":"evidence-benchmark-glycanml-bert-immunogenicity-bert-yaml","kind":"source","name":"GlycanML/GlycanML configs/single_task/BERT/immunogenicity_BERT.yaml","description":"Pinned computational evaluator or dataset/configuration source inspected for benchmark metadata.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://raw.githubusercontent.com/GlycanML/GlycanML/9f392aa6f9c6d74a296a250199beb347923d04e0/configs/single_task/BERT/immunogenicity_BERT.yaml","artifact_url":"https://raw.githubusercontent.com/GlycanML/GlycanML/9f392aa6f9c6d74a296a250199beb347923d04e0/configs/single_task/BERT/immunogenicity_BERT.yaml","artifact_sha256":"5c92f79c530629642450ca4536cd864cccc3a626f8ecdfeaaf6fcd2ece356901","version":"9f392aa6f9c6d74a296a250199beb347923d04e0","retrieved_at":"2026-09-16T20:42:46.930612+00:00"}} {"id":"evidence-benchmark-glycanml-bert-interaction-bert-yaml","kind":"source","name":"GlycanML/GlycanML configs/single_task/BERT/interaction_BERT.yaml","description":"Pinned computational evaluator or dataset/configuration source inspected for benchmark metadata.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://raw.githubusercontent.com/GlycanML/GlycanML/9f392aa6f9c6d74a296a250199beb347923d04e0/configs/single_task/BERT/interaction_BERT.yaml","artifact_url":"https://raw.githubusercontent.com/GlycanML/GlycanML/9f392aa6f9c6d74a296a250199beb347923d04e0/configs/single_task/BERT/interaction_BERT.yaml","artifact_sha256":"7cb8f9e774ab0f9d66536b2db47bb236214275ee08d4de95540e4e51d0129639","version":"9f392aa6f9c6d74a296a250199beb347923d04e0","retrieved_at":"2026-09-16T20:42:47.417430+00:00"}} {"id":"evidence-benchmark-glycanml-bert-link-bert-yaml","kind":"source","name":"GlycanML/GlycanML configs/single_task/BERT/link_BERT.yaml","description":"Pinned computational evaluator or dataset/configuration source inspected for benchmark metadata.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://raw.githubusercontent.com/GlycanML/GlycanML/9f392aa6f9c6d74a296a250199beb347923d04e0/configs/single_task/BERT/link_BERT.yaml","artifact_url":"https://raw.githubusercontent.com/GlycanML/GlycanML/9f392aa6f9c6d74a296a250199beb347923d04e0/configs/single_task/BERT/link_BERT.yaml","artifact_sha256":"82f43f0d45085132ecd6e6dff1b95bf7eb20ad5650713fc8ad0a42d0bbba5a94","version":"9f392aa6f9c6d74a296a250199beb347923d04e0","retrieved_at":"2026-09-16T20:42:47.161322+00:00"}} {"id":"evidence-benchmark-glycanml-bert-species-bert-yaml","kind":"source","name":"GlycanML/GlycanML configs/single_task/BERT/species_BERT.yaml","description":"Pinned computational evaluator or dataset/configuration source inspected for benchmark metadata.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://raw.githubusercontent.com/GlycanML/GlycanML/9f392aa6f9c6d74a296a250199beb347923d04e0/configs/single_task/BERT/species_BERT.yaml","artifact_url":"https://raw.githubusercontent.com/GlycanML/GlycanML/9f392aa6f9c6d74a296a250199beb347923d04e0/configs/single_task/BERT/species_BERT.yaml","artifact_sha256":"474995bb150c76479b0c99b6f0b2de49f5b6ad626eae1751e2f53774a4fbf38e","version":"9f392aa6f9c6d74a296a250199beb347923d04e0","retrieved_at":"2026-09-16T20:42:46.675170+00:00"}} {"id":"evidence-benchmark-massspecgym-de-novo-base-py","kind":"source","name":"pluskal-lab/MassSpecGym massspecgym/models/de_novo/base.py","description":"Pinned computational evaluator or dataset/configuration source inspected for benchmark metadata.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://raw.githubusercontent.com/pluskal-lab/MassSpecGym/f259fe3780d5bd227fc6ece36ce6f397c2eef716/massspecgym/models/de_novo/base.py","artifact_url":"https://raw.githubusercontent.com/pluskal-lab/MassSpecGym/f259fe3780d5bd227fc6ece36ce6f397c2eef716/massspecgym/models/de_novo/base.py","artifact_sha256":"5d2d61d7ac3c6503b3869d5cec687e7e82d438900df3d01c6f7cf4c340e5a004","version":"f259fe3780d5bd227fc6ece36ce6f397c2eef716","retrieved_at":"2026-09-16T20:42:47.661606+00:00"}} {"id":"evidence-benchmark-massspecgym-retrieval-base-py","kind":"source","name":"pluskal-lab/MassSpecGym massspecgym/models/retrieval/base.py","description":"Pinned computational evaluator or dataset/configuration source inspected for benchmark metadata.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://raw.githubusercontent.com/pluskal-lab/MassSpecGym/f259fe3780d5bd227fc6ece36ce6f397c2eef716/massspecgym/models/retrieval/base.py","artifact_url":"https://raw.githubusercontent.com/pluskal-lab/MassSpecGym/f259fe3780d5bd227fc6ece36ce6f397c2eef716/massspecgym/models/retrieval/base.py","artifact_sha256":"2ae08f58bd6db0e00430b54c078b45ae91b9252aac02ee358fef9d8ba1b3f81f","version":"f259fe3780d5bd227fc6ece36ce6f397c2eef716","retrieved_at":"2026-09-16T20:42:47.889112+00:00"}} {"id":"evidence-benchmark-massspecgym-simulation-base-py","kind":"source","name":"pluskal-lab/MassSpecGym massspecgym/models/simulation/base.py","description":"Pinned computational evaluator or dataset/configuration source inspected for benchmark metadata.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://raw.githubusercontent.com/pluskal-lab/MassSpecGym/f259fe3780d5bd227fc6ece36ce6f397c2eef716/massspecgym/models/simulation/base.py","artifact_url":"https://raw.githubusercontent.com/pluskal-lab/MassSpecGym/f259fe3780d5bd227fc6ece36ce6f397c2eef716/massspecgym/models/simulation/base.py","artifact_sha256":"5208cc352c144e24ea8249124483856aeb1e9821c0cea142d6a907ac248b6295","version":"f259fe3780d5bd227fc6ece36ce6f397c2eef716","retrieved_at":"2026-09-16T20:42:48.138101+00:00"}} {"id":"evidence-benchmark-mfass-pinned-readme","kind":"source","name":"MFASS computational protocol README at bee9133b83f3aedaf2bbb9013f1875515845607e","description":"","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/timini/rewire-benchmarks/blob/bee9133b83f3aedaf2bbb9013f1875515845607e/benchmarks/mfass/README.md","version":"bee9133b83f3aedaf2bbb9013f1875515845607e","retrieved_at":"2026-09-16T20:55:04.446365+00:00","artifact_sha256":"62a9381484fd2360767e78b114d9aa6cd8a4fc6487a981881a4c4999f0ab1b23","artifact_url":"https://raw.githubusercontent.com/timini/rewire-benchmarks/bee9133b83f3aedaf2bbb9013f1875515845607e/benchmarks/mfass/README.md"}} {"id":"evidence-benchmark-proteinbench-snapshot","kind":"source","name":"proteinbench official source","description":"Primary project documentation or project-maintained evidence.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"2e488850a6557bb57407615f2df9194351718b3dc0298a03c0c97d8e93460712","licence":null,"locator":"HTML main page","missing_metadata":{"immutable_source_version":"unavailable","licence":"unextracted","version":"unreported"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:31:54.349409+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","scope_note":"Mutable official page: content hash and retrieval time recorded, but the original page is not redistributed.","url":"https://proteinbench.github.io/","version":"Retrieved website snapshot sha256:2e488850a6557bb57407615f2df9194351718b3dc0298a03c0c97d8e93460712","artifact_url":"https://proteinbench.github.io/"}} {"id":"evidence-benchmark-structure-informed-html","kind":"source","name":"Structure-Informed Protein Language Models are Robust Predictors for Variant Effects (reviewed HTML snapshot)","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"6686d2646e7b1b1203a7ef6d48bacf966b4cd6b0895fd2855551937c6bb87111","artifact_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12068927/","retrieved_at":"2026-09-16T19:58:11.209175+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12068927/","version":"Human Genetics 2025 journal article (online 2024)"}} {"id":"evidence-benchmark-vcc2026-snapshot","kind":"source","name":"vcc2026 official source","description":"Primary project documentation or project-maintained evidence.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"57dd9cbd5e6b62ed665eebd909d82ffec580a55c3bd2f48a928f807a8c983b1e","licence":null,"locator":"HTML main page","missing_metadata":{"immutable_source_version":"unavailable","licence":"unextracted","version":"unreported"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:31:54.531595+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","scope_note":"Mutable official page: content hash and retrieval time recorded, but the original page is not redistributed.","url":"https://arcinstitute.org/news/virtual-cell-challenge-2026","version":"Retrieved website snapshot sha256:57dd9cbd5e6b62ed665eebd909d82ffec580a55c3bd2f48a928f807a8c983b1e","artifact_url":"https://arcinstitute.org/news/virtual-cell-challenge-2026"}} {"id":"evidence-discovery-final-amber","kind":"source","name":"amber primary benchmark evidence","description":"Primary paper or official task implementation inspected for the named methodology claims.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"2a75501cbe44396ec0d103d896026d19d823336d131e1fa2aada8555f495a7d0","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC6022608/fullTextXML","locator":"Full primary artifact; individual claims specify their section.","retrieved_at":"2026-09-16T21:04:55.738801+00:00","review_method":"automated_source_review","review_scope":"Benchmark methodology metadata only; no independent reproduction or new numerical result extraction.","url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC6022608/fullTextXML","version":"PMC6022608"}} {"id":"evidence-discovery-final-atom3d","kind":"source","name":"atom3d primary benchmark evidence","description":"Primary paper or official task implementation inspected for the named methodology claims.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"92656c20a15311c32bed9edc7f465bb26eb30338f40324fe43bec4b1fc6a7890","artifact_url":"https://arxiv.org/pdf/2012.04035v4","locator":"Full primary artifact; individual claims specify their section.","retrieved_at":"2026-09-16T21:05:49.738972+00:00","review_method":"automated_source_review","review_scope":"Benchmark methodology metadata only; no independent reproduction or new numerical result extraction.","url":"https://arxiv.org/abs/2012.04035v4","version":"arXiv:2012.04035v4 (15 January 2022)"}} {"id":"evidence-discovery-final-beacon","kind":"source","name":"beacon primary benchmark evidence","description":"Primary paper or official task implementation inspected for the named methodology claims.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"c2496bff164b87e635ba253ea5a5edc55c94c4673e81c071bc8f0db1e288d006","artifact_url":"https://arxiv.org/pdf/2406.10391v1","locator":"Full primary artifact; individual claims specify their section.","retrieved_at":"2026-09-16T21:04:55.172530+00:00","review_method":"automated_source_review","review_scope":"Benchmark methodology metadata only; no independent reproduction or new numerical result extraction.","url":"https://arxiv.org/pdf/2406.10391v1","version":"2406.10391v1"}} {"id":"evidence-discovery-final-beeline","kind":"source","name":"beeline primary benchmark evidence","description":"Primary paper or official task implementation inspected for the named methodology claims.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"99b59a6941878779b34ab8eed9ea27c8d8967c24fb67b92fce418e7f512d3a11","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC7098173/fullTextXML","locator":"Full primary artifact; individual claims specify their section.","retrieved_at":"2026-09-16T21:04:55.226418+00:00","review_method":"automated_source_review","review_scope":"Benchmark methodology metadata only; no independent reproduction or new numerical result extraction.","url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC7098173/fullTextXML","version":"PMC7098173"}} {"id":"evidence-discovery-final-bend","kind":"source","name":"bend primary benchmark evidence","description":"Primary paper or official task implementation inspected for the named methodology claims.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"6ff9f6dc19fc831200e241e3279a06c44d0fcd0aac566da81058e9db7fb70334","artifact_url":"https://arxiv.org/pdf/2311.12570v1","locator":"Full primary artifact; individual claims specify their section.","retrieved_at":"2026-09-16T21:04:56.127991+00:00","review_method":"automated_source_review","review_scope":"Benchmark methodology metadata only; no independent reproduction or new numerical result extraction.","url":"https://arxiv.org/pdf/2311.12570v1","version":"2311.12570v1"}} {"id":"evidence-discovery-final-cafa3","kind":"source","name":"cafa3 primary benchmark evidence","description":"Primary paper or official task implementation inspected for the named methodology claims.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"3802f37548d77addac829127bf03322cbe6c69eba944d88a775b534311804fa4","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC6864930/fullTextXML","locator":"Full primary artifact; individual claims specify their section.","retrieved_at":"2026-09-16T21:05:50.876204+00:00","review_method":"automated_source_review","review_scope":"Benchmark methodology metadata only; no independent reproduction or new numerical result extraction.","url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC6864930/fullTextXML","version":"PMC6864930"}} {"id":"evidence-discovery-final-cami2","kind":"source","name":"cami2 primary benchmark evidence","description":"Primary paper or official task implementation inspected for the named methodology claims.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"da932c1cde8b290e1694fc3cf44d98ed527be1ec7cf41542dd5e9aee75c38704","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC9007738/fullTextXML","locator":"Full primary artifact; individual claims specify their section.","retrieved_at":"2026-09-16T21:04:55.691966+00:00","review_method":"automated_source_review","review_scope":"Benchmark methodology metadata only; no independent reproduction or new numerical result extraction.","url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC9007738/fullTextXML","version":"PMC9007738"}} {"id":"evidence-discovery-final-capri","kind":"source","name":"capri primary benchmark evidence","description":"Primary paper or official task implementation inspected for the named methodology claims.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"194e32f06903caee54706cb61570cb1d5bc6b47911a8a427d9244789ce5a0165","artifact_url":"https://www.capri-docking.org/assessment/","locator":"Full primary artifact; individual claims specify their section.","retrieved_at":"2026-09-16T21:11:45.067183+00:00","review_method":"automated_source_review","review_scope":"Benchmark methodology metadata only; no independent reproduction or new numerical result extraction.","url":"https://www.capri-docking.org/assessment/","version":"Retrieved 2026-09-16 assessment page"}} {"id":"evidence-discovery-final-casp16","kind":"source","name":"casp16 primary benchmark evidence","description":"Primary paper or official task implementation inspected for the named methodology claims.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"32ef3bdfac658a0191e2939699b68e23dcaf465568c5d1c9f75743e886bd76cc","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12750037/fullTextXML","locator":"Full primary artifact; individual claims specify their section.","retrieved_at":"2026-09-16T21:11:45.148220+00:00","review_method":"automated_source_review","review_scope":"Benchmark methodology metadata only; no independent reproduction or new numerical result extraction.","url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12750037/fullTextXML","version":"PMC12750037"}} {"id":"evidence-discovery-final-dart","kind":"source","name":"dart primary benchmark evidence","description":"Primary paper or official task implementation inspected for the named methodology claims.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"4194b137ba55c9a2c269d119a9afec6ae1bb0feaf17d91433ae483c41221a56b","artifact_url":"https://arxiv.org/html/2412.05430v1","locator":"Full primary artifact; individual claims specify their section.","retrieved_at":"2026-09-16T21:06:29.715960+00:00","review_method":"automated_source_review","review_scope":"Benchmark methodology metadata only; no independent reproduction or new numerical result extraction.","url":"https://arxiv.org/html/2412.05430v1","version":"2412.05430v1"}} {"id":"evidence-discovery-final-dockq","kind":"source","name":"dockq primary benchmark evidence","description":"Primary paper or official task implementation inspected for the named methodology claims.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"013ceed0e054f971a81caf7cdcac10c3f4241ecec0532e326a1fca124e69dbe9","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC4999177/fullTextXML","locator":"Full primary artifact; individual claims specify their section.","retrieved_at":"2026-09-16T21:11:45.113639+00:00","review_method":"automated_source_review","review_scope":"Benchmark methodology metadata only; no independent reproduction or new numerical result extraction.","url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC4999177/fullTextXML","version":"PMC4999177"}} {"id":"evidence-discovery-final-flip","kind":"source","name":"flip primary benchmark evidence","description":"Primary paper or official task implementation inspected for the named methodology claims.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"afcf360c88a7a4ae153b3c2d8d4fa6d4ac0abe84f2131b94409abff6447ce363","artifact_url":"https://flip.protein.properties/assets/FLIP_2021_manuscript.pdf","locator":"Full primary artifact; individual claims specify their section.","retrieved_at":"2026-09-16T21:07:13.889041+00:00","review_method":"automated_source_review","review_scope":"Benchmark methodology metadata only; no independent reproduction or new numerical result extraction.","url":"https://flip.protein.properties/assets/FLIP_2021_manuscript.pdf","version":"2021 manuscript"}} {"id":"evidence-discovery-final-flip2","kind":"source","name":"flip2 primary benchmark evidence","description":"Primary paper or official task implementation inspected for the named methodology claims.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"d0e61ca27863c023adddcc225a3e4af1718f2e3fdfe3642bef89c24c93621a17","artifact_url":"https://flip.protein.properties/assets/FLIP_manuscipt.pdf","locator":"Full primary artifact; individual claims specify their section.","retrieved_at":"2026-09-16T21:05:54.225877+00:00","review_method":"automated_source_review","review_scope":"Benchmark methodology metadata only; no independent reproduction or new numerical result extraction.","url":"https://flip.protein.properties/assets/FLIP_manuscipt.pdf","version":"ICML2026 manuscript"}} {"id":"evidence-discovery-final-geneb","kind":"source","name":"geneb primary benchmark evidence","description":"Primary paper or official task implementation inspected for the named methodology claims.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"47975089c0ca738d5e2d6e6ea91cd4e7b80498c3f175803d8ec39ca77aa41629","artifact_url":"https://arxiv.org/html/2606.04525v1","locator":"Full primary artifact; individual claims specify their section.","retrieved_at":"2026-09-16T21:06:29.746844+00:00","review_method":"automated_source_review","review_scope":"Benchmark methodology metadata only; no independent reproduction or new numerical result extraction.","url":"https://arxiv.org/html/2606.04525v1","version":"2606.04525v1"}} {"id":"evidence-discovery-final-genomic-benchmarks","kind":"source","name":"genomic-benchmarks primary benchmark evidence","description":"Primary paper or official task implementation inspected for the named methodology claims.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"bda6fe51e3363a5d2fc8d265ca536897d3e83eb76458fc95066c21933e3bd0c0","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC10150520/fullTextXML","locator":"Full primary artifact; individual claims specify their section.","retrieved_at":"2026-09-16T21:07:13.681402+00:00","review_method":"automated_source_review","review_scope":"Benchmark methodology metadata only; no independent reproduction or new numerical result extraction.","url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC10150520/fullTextXML","version":"PMC10150520"}} {"id":"evidence-discovery-final-glycanml","kind":"source","name":"glycanml primary benchmark evidence","description":"Primary paper or official task implementation inspected for the named methodology claims.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"9ba3db678b4550898a612b42e8832bf9dee40935090fc4d969ba8f6ac5106979","artifact_url":"https://arxiv.org/pdf/2405.16206v1","locator":"Full primary artifact; individual claims specify their section.","retrieved_at":"2026-09-16T21:04:56.017006+00:00","review_method":"automated_source_review","review_scope":"Benchmark methodology metadata only; no independent reproduction or new numerical result extraction.","url":"https://arxiv.org/pdf/2405.16206v1","version":"2405.16206v1"}} {"id":"evidence-discovery-final-gue","kind":"source","name":"gue primary benchmark evidence","description":"Primary paper or official task implementation inspected for the named methodology claims.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"49300acee3e4afd44bebc3de9893c3bc310d331bd4805374e0952fdfbf366f06","artifact_url":"https://arxiv.org/pdf/2306.15006","locator":"Full primary artifact; individual claims specify their section.","retrieved_at":"2026-09-16T20:00:00+00:00","review_method":"automated_source_review","review_scope":"Benchmark methodology metadata only; no independent reproduction or new numerical result extraction.","url":"https://arxiv.org/pdf/2306.15006","version":"2306.15006 retrieved PDF"}} {"id":"evidence-discovery-final-hest","kind":"source","name":"hest primary benchmark evidence","description":"Primary paper or official task implementation inspected for the named methodology claims.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"636099a73dee8337f60e6e9120230b914605b35553872ddf76e4661bbe14be9b","artifact_url":"https://arxiv.org/pdf/2406.16192v1","locator":"Full primary artifact; individual claims specify their section.","retrieved_at":"2026-09-16T21:05:00.540640+00:00","review_method":"automated_source_review","review_scope":"Benchmark methodology metadata only; no independent reproduction or new numerical result extraction.","url":"https://arxiv.org/pdf/2406.16192v1","version":"2406.16192v1"}} {"id":"evidence-discovery-final-massspecgym","kind":"source","name":"massspecgym primary benchmark evidence","description":"Primary paper or official task implementation inspected for the named methodology claims.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"82176d50e8947c8b9baa2a0d91493f5680c0e4c7e25a2266ff7879f48a58c40c","artifact_url":"https://arxiv.org/pdf/2410.23326v1","locator":"Full primary artifact; individual claims specify their section.","retrieved_at":"2026-09-16T21:04:58.775040+00:00","review_method":"automated_source_review","review_scope":"Benchmark methodology metadata only; no independent reproduction or new numerical result extraction.","url":"https://arxiv.org/pdf/2410.23326v1","version":"2410.23326v1"}} {"id":"evidence-discovery-final-mrnabench","kind":"source","name":"mrnabench primary benchmark evidence","description":"Primary paper or official task implementation inspected for the named methodology claims.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"79f6264ee883535203c63a313547e7c57baa85585f76b42f8d899eb17fb7e600","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12265608/fullTextXML","locator":"Full primary artifact; individual claims specify their section.","retrieved_at":"2026-09-16T10:41:16.497221+00:00","review_method":"automated_source_review","review_scope":"Benchmark methodology metadata only; no independent reproduction or new numerical result extraction.","url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12265608/fullTextXML","version":"preprint archived 2025-07-08"}} {"id":"evidence-discovery-final-nabench","kind":"source","name":"nabench primary benchmark evidence","description":"Primary paper or official task implementation inspected for the named methodology claims.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"fefd48d53b1a7eadf9e14db96adacc8e646304c1b592562d9f136f4508941350","artifact_url":"https://arxiv.org/html/2511.02888v1","locator":"Full primary artifact; individual claims specify their section.","retrieved_at":"2026-09-16T21:06:29.579095+00:00","review_method":"automated_source_review","review_scope":"Benchmark methodology metadata only; no independent reproduction or new numerical result extraction.","url":"https://arxiv.org/html/2511.02888v1","version":"2511.02888v1"}} {"id":"evidence-discovery-final-opal","kind":"source","name":"opal primary benchmark evidence","description":"Primary paper or official task implementation inspected for the named methodology claims.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"a4503ef32e99127d55d0af8ebf0c39baa298b320e4032fcacdb9aa5d33821a09","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC6398228/fullTextXML","locator":"Full primary artifact; individual claims specify their section.","retrieved_at":"2026-09-16T21:04:56.586502+00:00","review_method":"automated_source_review","review_scope":"Benchmark methodology metadata only; no independent reproduction or new numerical result extraction.","url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC6398228/fullTextXML","version":"PMC6398228"}} {"id":"evidence-discovery-final-openproblems-label","kind":"source","name":"openproblems-label primary benchmark evidence","description":"Primary paper or official task implementation inspected for the named methodology claims.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"e223ab712ff55997a3abe659f280d4ea2952700e767b87e02c434701da9833c1","artifact_url":"https://www.openproblems.bio/benchmarks/label_projection/v1.0.0/","locator":"Full primary artifact; individual claims specify their section.","retrieved_at":"2026-09-16T21:16:30.026457+00:00","review_method":"automated_source_review","review_scope":"Benchmark methodology metadata only; no independent reproduction or new numerical result extraction.","url":"https://www.openproblems.bio/benchmarks/label_projection/v1.0.0/","version":"v1.0.0"}} {"id":"evidence-discovery-final-perturbench","kind":"source","name":"perturbench primary benchmark evidence","description":"Primary paper or official task implementation inspected for the named methodology claims.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"5c4804565dd9faa17a11853a79e9847dcb6da73715b4c91f62e5b274cc79f186","artifact_url":"https://arxiv.org/html/2408.10609v1","locator":"Full primary artifact; individual claims specify their section.","retrieved_at":"2026-09-16T21:07:13.159027+00:00","review_method":"automated_source_review","review_scope":"Benchmark methodology metadata only; no independent reproduction or new numerical result extraction.","url":"https://arxiv.org/html/2408.10609v1","version":"2408.10609v1"}} {"id":"evidence-discovery-final-petab","kind":"source","name":"petab primary benchmark evidence","description":"Primary paper or official task implementation inspected for the named methodology claims.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"9a9f73410331cb6d0ee148f6872f191393cbaa334f3437ed7fd0fe289476e7a7","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC6735869/fullTextXML","locator":"Full primary artifact; individual claims specify their section.","retrieved_at":"2026-09-16T21:04:56.573248+00:00","review_method":"automated_source_review","review_scope":"Benchmark methodology metadata only; no independent reproduction or new numerical result extraction.","url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC6735869/fullTextXML","version":"PMC6735869"}} {"id":"evidence-discovery-final-pfmbench","kind":"source","name":"pfmbench primary benchmark evidence","description":"Primary paper or official task implementation inspected for the named methodology claims.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"59c7bbb888e8e91f33c1e2cabfde062c32381d6ca727d23b6655c977aabf97a2","artifact_url":"https://arxiv.org/html/2506.14796v1","locator":"Full primary artifact; individual claims specify their section.","retrieved_at":"2026-09-16T21:06:29.716004+00:00","review_method":"automated_source_review","review_scope":"Benchmark methodology metadata only; no independent reproduction or new numerical result extraction.","url":"https://arxiv.org/html/2506.14796v1","version":"2506.14796v1"}} {"id":"evidence-discovery-final-plinder-config0","kind":"source","name":"plinder-config0 primary benchmark evidence","description":"Primary paper or official task implementation inspected for the named methodology claims.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"23ad37a97187942840416e240d7ee99d0a677a4964a75beed3e8be0082b9375d","artifact_url":"https://raw.githubusercontent.com/plinder-org/plinder/85b3f1cb1763530a6cfd934f4263a1777c41afa4/docs/evaluation.md","locator":"Full primary artifact; individual claims specify their section.","retrieved_at":"2026-09-16T21:11:45.036654+00:00","review_method":"automated_source_review","review_scope":"Benchmark methodology metadata only; no independent reproduction or new numerical result extraction.","url":"https://raw.githubusercontent.com/plinder-org/plinder/85b3f1cb1763530a6cfd934f4263a1777c41afa4/docs/evaluation.md","version":"85b3f1cb1763530a6cfd934f4263a1777c41afa4:docs/evaluation.md"}} {"id":"evidence-discovery-final-plinder-readme","kind":"source","name":"plinder-readme primary benchmark evidence","description":"Primary paper or official task implementation inspected for the named methodology claims.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"1d53c3b89030dc4651d3e7bf4749256b7e579330fc7a660cffaa992d646da34a","artifact_url":"https://github.com/plinder-org/plinder/blob/85b3f1cb1763530a6cfd934f4263a1777c41afa4/README.md","locator":"Full primary artifact; individual claims specify their section.","retrieved_at":"2026-09-16T10:30:23.659162+00:00","review_method":"automated_source_review","review_scope":"Benchmark methodology metadata only; no independent reproduction or new numerical result extraction.","url":"https://github.com/plinder-org/plinder/blob/85b3f1cb1763530a6cfd934f4263a1777c41afa4/README.md","version":"85b3f1cb1763530a6cfd934f4263a1777c41afa4"}} {"id":"evidence-discovery-final-posebusters","kind":"source","name":"posebusters primary benchmark evidence","description":"Primary paper or official task implementation inspected for the named methodology claims.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"8bbc6eadc59d33f7433b610d91c0b0d2d3094cacb3c0bdfa83bf89239580c763","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC10901501/fullTextXML","locator":"Full primary artifact; individual claims specify their section.","retrieved_at":"2026-09-16T21:05:51.089997+00:00","review_method":"automated_source_review","review_scope":"Benchmark methodology metadata only; no independent reproduction or new numerical result extraction.","url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC10901501/fullTextXML","version":"PMC10901501"}} {"id":"evidence-discovery-final-proteinbench","kind":"source","name":"proteinbench primary benchmark evidence","description":"Primary paper or official task implementation inspected for the named methodology claims.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"4334d636223ad42bfb9ae68aae03f5a255c29ba1cebe7b8f9588fb3b9b5453b2","artifact_url":"https://arxiv.org/html/2409.06744v1","locator":"Full primary artifact; individual claims specify their section.","retrieved_at":"2026-09-16T21:07:13.231727+00:00","review_method":"automated_source_review","review_scope":"Benchmark methodology metadata only; no independent reproduction or new numerical result extraction.","url":"https://arxiv.org/html/2409.06744v1","version":"2409.06744v1"}} {"id":"evidence-discovery-final-proteingym","kind":"source","name":"proteingym primary benchmark evidence","description":"Primary paper or official task implementation inspected for the named methodology claims.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"4519641f13271bdd09b166e7d93232f22542489bc52a25b5a1628c3df8badce1","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC10723403/fullTextXML","locator":"Full primary artifact; individual claims specify their section.","retrieved_at":"2026-09-16T10:41:16.517323+00:00","review_method":"automated_source_review","review_scope":"Benchmark methodology metadata only; no independent reproduction or new numerical result extraction.","url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC10723403/fullTextXML","version":"PMC10723403.1"}} {"id":"evidence-discovery-final-scib","kind":"source","name":"scib primary benchmark evidence","description":"Primary paper or official task implementation inspected for the named methodology claims.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"f65dd8b63336ff1a5045dad3cb9c5905ec3bb8b494a221f67dfbffb9a6f612db","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC8748196/fullTextXML","locator":"Full primary artifact; individual claims specify their section.","retrieved_at":"2026-09-16T21:04:56.581902+00:00","review_method":"automated_source_review","review_scope":"Benchmark methodology metadata only; no independent reproduction or new numerical result extraction.","url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC8748196/fullTextXML","version":"PMC8748196"}} {"id":"evidence-discovery-final-scperteval0","kind":"source","name":"scperteval0 primary benchmark evidence","description":"Primary paper or official task implementation inspected for the named methodology claims.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"c4dbfdbc539ba78350ddba68ca4c03edfe89cf3625c6c699884bd3a85e47d929","artifact_url":"https://raw.githubusercontent.com/Virtual-Cell-Research-Community/scPertEval/4685f11927e887745737600170da7a655b727553/src/scperteval/protocols/metrics.py","locator":"Full primary artifact; individual claims specify their section.","retrieved_at":"2026-09-16T21:11:45.124583+00:00","review_method":"automated_source_review","review_scope":"Benchmark methodology metadata only; no independent reproduction or new numerical result extraction.","url":"https://raw.githubusercontent.com/Virtual-Cell-Research-Community/scPertEval/4685f11927e887745737600170da7a655b727553/src/scperteval/protocols/metrics.py","version":"4685f11927e887745737600170da7a655b727553:src/scperteval/protocols/metrics.py"}} {"id":"evidence-discovery-final-tape","kind":"source","name":"tape primary benchmark evidence","description":"Primary paper or official task implementation inspected for the named methodology claims.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"6ee0c3e6e870635cba8fa67e0a4abc5598c0ab2a10ba127a46b67ac450ae0168","artifact_url":"https://arxiv.org/pdf/1906.08230v1","locator":"Full primary artifact; individual claims specify their section.","retrieved_at":"2026-09-16T20:23:48.238777+00:00","review_method":"automated_source_review","review_scope":"Benchmark methodology metadata only; no independent reproduction or new numerical result extraction.","url":"https://arxiv.org/pdf/1906.08230v1","version":"1906.08230v1"}} {"id":"evidence-discovery-final-tdc","kind":"source","name":"tdc primary benchmark evidence","description":"Primary paper or official task implementation inspected for the named methodology claims.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"aaa5526f6f100093bc06c9921a8564ad08cfd5332270d614d7681051c2dd66bc","artifact_url":"https://arxiv.org/pdf/2102.09548v1","locator":"Full primary artifact; individual claims specify their section.","retrieved_at":"2026-09-16T21:04:55.709744+00:00","review_method":"automated_source_review","review_scope":"Benchmark methodology metadata only; no independent reproduction or new numerical result extraction.","url":"https://arxiv.org/pdf/2102.09548v1","version":"2102.09548v1"}} {"id":"evidence-discovery-final-vcc-guide","kind":"source","name":"vcc-guide primary benchmark evidence","description":"Primary paper or official task implementation inspected for the named methodology claims.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"a2a085fd104ecf6b97e125ebc5714c642e8140db4f549ff84abd3bce1d2a2c80","artifact_url":"https://vcc-cli-wiki.virtualcellchallenge.org/","locator":"Full primary artifact; individual claims specify their section.","retrieved_at":"2026-09-16T21:16:29.975126+00:00","review_method":"automated_source_review","review_scope":"Benchmark methodology metadata only; no independent reproduction or new numerical result extraction.","url":"https://vcc-cli-wiki.virtualcellchallenge.org/","version":"Retrieved 2026-09-16"}} {"id":"evidence-final-model-geneformer-tree","kind":"source","name":"Geneformer official repository file inventory","description":"Primary-source follow-up resolving earlier evidence retrieval gaps.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://huggingface.co/ctheodoris/Geneformer/tree/1f7fbae4e469a5f4f1af8c111a529cfe1b3829f5","artifact_url":"https://huggingface.co/ctheodoris/Geneformer/tree/1f7fbae4e469a5f4f1af8c111a529cfe1b3829f5","version":"1f7fbae4e469a5f4f1af8c111a529cfe1b3829f5","artifact_sha256":"d1e85b25927c06241c2240f8272223838ab279f5072ae0a270a34efc8185d7dc","artifact_format":"extracted_file_path_inventory","retrieved_at":"2026-09-16T21:10:17.420034+00:00"}} {"id":"evidence-final-model-rna-fm-paper","kind":"source","name":"RNA-FM original methods, arXiv v5","description":"Primary-source follow-up resolving earlier evidence retrieval gaps.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://arxiv.org/pdf/2204.00300v5","artifact_url":"https://arxiv.org/pdf/2204.00300v5","version":"2204.00300v5","artifact_sha256":"b3945c283c5ca6e8a346cfeb3e8d6609dac6e4a3874fd38dc2436c35eeb6074a","artifact_format":"original_pdf","retrieved_at":"2026-09-16T21:10:17.419803+00:00"}} {"id":"evidence-mfass-final-geo","kind":"source","name":"MFASS original assay accession GSE120695","description":"Primary artifact inspected for profile metadata; historical numerical results are not changed.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://ftp.ncbi.nlm.nih.gov/geo/series/GSE120nnn/GSE120695/soft/GSE120695_family.soft.gz","artifact_url":"https://ftp.ncbi.nlm.nih.gov/geo/series/GSE120nnn/GSE120695/soft/GSE120695_family.soft.gz","artifact_sha256":"a85331f962b962747ba39c8d0518b0f14d90ca8db3d8396f8fe2f2ffc3e48fa3","version":"GSE120695 SOFT retrieved 2026-09-16","retrieved_at":"2026-09-16T21:02:58.608027+00:00"}} {"id":"evidence-mfass-final-v1-comparison","kind":"source","name":"Archived MFASS v1 SpliceAI comparison","description":"Primary artifact inspected for profile metadata; historical numerical results are not changed.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/bee9133b83f3aedaf2bbb9013f1875515845607e/benchmarks/mfass/results/compare-baseline-vs-spliceai.json","artifact_url":"https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/bee9133b83f3aedaf2bbb9013f1875515845607e/benchmarks/mfass/results/compare-baseline-vs-spliceai.json","artifact_sha256":"4856394ecb59c7163afe5d3ef50d5ab9c2cec5b20ed3b4f3b39ef8dfc02b5778","version":"bee9133b83f3aedaf2bbb9013f1875515845607e","retrieved_at":"2026-09-16T21:02:58.608188+00:00"}} {"id":"evidence-official-00ee97a5586392ae2348","kind":"source","name":"soedinglab/hh-suite: LICENSE","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/soedinglab/hh-suite/blob/43095e46ada4ec2a8a47d47ef5ad7e38b1429f7b/LICENSE","artifact_url":"https://raw.githubusercontent.com/soedinglab/hh-suite/43095e46ada4ec2a8a47d47ef5ad7e38b1429f7b/LICENSE","version":"43095e46ada4ec2a8a47d47ef5ad7e38b1429f7b","retrieved_at":"2026-09-16T19:46:20.402052+00:00","artifact_sha256":"589ed823e9a84c56feb95ac58e7cf384626b9cbf4fda2a907bc36e103de1bad2","locator":"LICENSE","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-019aff235ae2c3cf29d6","kind":"source","name":"dauparas/ProteinMPNN: protein_mpnn_utils.py","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/dauparas/ProteinMPNN/blob/8907e6671bfbfc92303b5f79c4b5e6ce47cdef57/protein_mpnn_utils.py","artifact_url":"https://raw.githubusercontent.com/dauparas/ProteinMPNN/8907e6671bfbfc92303b5f79c4b5e6ce47cdef57/protein_mpnn_utils.py","version":"8907e6671bfbfc92303b5f79c4b5e6ce47cdef57","retrieved_at":"2026-09-16T19:46:18.948060+00:00","artifact_sha256":"74c8f9b7553422a7a0bbd705874844ee103c8926c2c96f154a87e0b824071e1b","locator":"protein_mpnn_utils.py","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-028db5eeb224def23456","kind":"source","name":"songlab-cal/tape: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/songlab-cal/tape/blob/6d345c2b2bbf52cd32cf179325c222afd92aec7e/README.md","artifact_url":"https://raw.githubusercontent.com/songlab-cal/tape/6d345c2b2bbf52cd32cf179325c222afd92aec7e/README.md","version":"6d345c2b2bbf52cd32cf179325c222afd92aec7e","retrieved_at":"2026-09-16T19:46:20.570146+00:00","artifact_sha256":"b28c74fe3cd6b69a8ba6d84891d0539e54dfef882abd5ed4d11ed0b029bb477a","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-03106594cde7719aa359","kind":"source","name":"nbrg-ppcu/prokbert: LICENSE","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/nbrg-ppcu/prokbert/blob/8670ae92b816cff158a0b85647a8dea122e251eb/LICENSE","artifact_url":"https://raw.githubusercontent.com/nbrg-ppcu/prokbert/8670ae92b816cff158a0b85647a8dea122e251eb/LICENSE","version":"8670ae92b816cff158a0b85647a8dea122e251eb","retrieved_at":"2026-09-16T19:46:19.913949+00:00","artifact_sha256":"66e17f3a57ace034ab422667c02d75134bb531619d9491f3abc7ea42bb8a6643","locator":"LICENSE","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-0437be6150a3da9f8efc","kind":"source","name":"evolutionaryscale/esm: _assets/ESM3_README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/evolutionaryscale/esm/blob/bf343ba264b650dff7a073643725f9aaa1fdbe8d/_assets/ESM3_README.md","artifact_url":"https://raw.githubusercontent.com/evolutionaryscale/esm/bf343ba264b650dff7a073643725f9aaa1fdbe8d/_assets/ESM3_README.md","version":"bf343ba264b650dff7a073643725f9aaa1fdbe8d","retrieved_at":"2026-09-16T19:46:18.983347+00:00","artifact_sha256":"2f564153712e32ad668f1586b6e78599d1a7fd6a3b3b7049718d85f9b33f79a6","locator":"_assets/ESM3_README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-044ad0df4e4fe48a2125","kind":"source","name":"songlab-cal/tape: tape/models/modeling_unirep.py","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/songlab-cal/tape/blob/6d345c2b2bbf52cd32cf179325c222afd92aec7e/tape/models/modeling_unirep.py","artifact_url":"https://raw.githubusercontent.com/songlab-cal/tape/6d345c2b2bbf52cd32cf179325c222afd92aec7e/tape/models/modeling_unirep.py","version":"6d345c2b2bbf52cd32cf179325c222afd92aec7e","retrieved_at":"2026-09-16T19:46:20.570146+00:00","artifact_sha256":"f6a0da726a8bcdd40640fab39c0814a56d6188531b44a3ce6e7e7137755c9a02","locator":"tape/models/modeling_unirep.py","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-04ac37076f906b8470cb","kind":"source","name":"InstaDeepAI/agro-nucleotide-transformer-1b: config.json","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://huggingface.co/InstaDeepAI/agro-nucleotide-transformer-1b/blob/b0e1ea1f53a2bf5bb29f8eab7a7e553bf06c1ab1/config.json","artifact_url":"https://huggingface.co/InstaDeepAI/agro-nucleotide-transformer-1b/blob/b0e1ea1f53a2bf5bb29f8eab7a7e553bf06c1ab1/config.json","version":"b0e1ea1f53a2bf5bb29f8eab7a7e553bf06c1ab1","retrieved_at":"2026-09-16T20:12:20.913125+00:00","artifact_sha256":"4063f8250f32d922611d8b36f0def1bb53b7ae129d6439c50c8b6c340e8eb0bd","locator":"config.json","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-082d60e1af5af7ed12ca","kind":"source","name":"aqlaboratory/genie3: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/aqlaboratory/genie3/blob/d77ae5ac04212ff1e8b29b585859a3244c614804/README.md","artifact_url":"https://raw.githubusercontent.com/aqlaboratory/genie3/d77ae5ac04212ff1e8b29b585859a3244c614804/README.md","version":"d77ae5ac04212ff1e8b29b585859a3244c614804","retrieved_at":"2026-09-16T19:46:18.306022+00:00","artifact_sha256":"50a634fe3236b6272b70928ac41bdebd4a230467b151f59597742ccd56ac8909","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-099d109417c7fae96e96","kind":"source","name":"aertslab/GRNBoost: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/aertslab/GRNBoost/blob/26c836b3dcbb85852d3c6f4b8340e8655434da02/README.md","artifact_url":"https://raw.githubusercontent.com/aertslab/GRNBoost/26c836b3dcbb85852d3c6f4b8340e8655434da02/README.md","version":"26c836b3dcbb85852d3c6f4b8340e8655434da02","retrieved_at":"2026-09-16T19:46:18.179520+00:00","artifact_sha256":"f2befd99acf59576a22b8a44abd2345e8ed7304cf470609f27d311e08ed3f066","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-0b8003691827eb09b0a4","kind":"source","name":"proteinmpnn-supp: Publisher supplementary archive","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC9997061/supplementaryFiles","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC9997061/supplementaryFiles","version":"Retrieved page snapshot; no immutable publisher revision supplied","retrieved_at":"2026-09-16T20:43:11.978593+00:00","artifact_sha256":"002269144962619673b7b187bdda20bd250e5f3cfd9f72a529eac53e7a1ac082","locator":"Publisher supplementary archive","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-0cbab80c2d1769287a1f","kind":"source","name":"scverse/scvi-tools: docs/user_guide/models/scvi.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/scverse/scvi-tools/blob/73b28e44223621470e582a81a102c107bb22678b/docs/user_guide/models/scvi.md","artifact_url":"https://raw.githubusercontent.com/scverse/scvi-tools/73b28e44223621470e582a81a102c107bb22678b/docs/user_guide/models/scvi.md","version":"73b28e44223621470e582a81a102c107bb22678b","retrieved_at":"2026-09-16T19:46:20.137176+00:00","artifact_sha256":"c81a5cfb30db1eca8c03af3d0db1a36a447a1bb5a27b270de5633ba73ebeeeef","locator":"docs/user_guide/models/scvi.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-0e4dac85aa0e0ff1d4b2","kind":"source","name":"bowang-lab/scGPT: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/bowang-lab/scGPT/blob/cebd6fae655b9c585a4807daa3ac31bb764f06b4/README.md","artifact_url":"https://raw.githubusercontent.com/bowang-lab/scGPT/cebd6fae655b9c585a4807daa3ac31bb764f06b4/README.md","version":"cebd6fae655b9c585a4807daa3ac31bb764f06b4","retrieved_at":"2026-09-16T19:46:18.594429+00:00","artifact_sha256":"b0503e8ca789f19f1fc2350c5aaf57b1b323bbae43b354655231b5f4a1586c83","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-0ffe1031786a71400ffd","kind":"source","name":"mimic: Primary paper PDF","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://arxiv.org/pdf/2604.24506","artifact_url":"https://arxiv.org/pdf/2604.24506","version":"2604.24506v1","retrieved_at":"2026-09-16T20:04:04.776851+00:00","artifact_sha256":"0ba8639b263f0e7f736bbd4564a8b1acca1f66acc4300cffd3c26e6bf2e93ffe","locator":"Primary paper PDF","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-1005311cb2f598fe63de","kind":"source","name":"CAMI-challenge/CAMISIM: LICENSE.txt","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/CAMI-challenge/CAMISIM/blob/7ce6013c6d5a0fac8ba8a80e52a03560d3546fa6/LICENSE.txt","artifact_url":"https://raw.githubusercontent.com/CAMI-challenge/CAMISIM/7ce6013c6d5a0fac8ba8a80e52a03560d3546fa6/LICENSE.txt","version":"7ce6013c6d5a0fac8ba8a80e52a03560d3546fa6","retrieved_at":"2026-09-16T19:46:17.767933+00:00","artifact_sha256":"b40930bbcf80744c86c46a12bc9da056641d722716c378f5659b9e555ef833e1","locator":"LICENSE.txt","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-10a6550ca70fdeaf33cb","kind":"source","name":"matsui-lab/GlycanGT: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/matsui-lab/GlycanGT/blob/96611518c971deb89215ca163deaf9de3a59fa32/README.md","artifact_url":"https://raw.githubusercontent.com/matsui-lab/GlycanGT/96611518c971deb89215ca163deaf9de3a59fa32/README.md","version":"96611518c971deb89215ca163deaf9de3a59fa32","retrieved_at":"2026-09-16T19:46:19.737544+00:00","artifact_sha256":"fa39bffca31211baedb3b63df5dede775fcffff7b411d4261d5b18f526ae1153","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-10f58e320879def8b102","kind":"source","name":"biobakery/humann: readme.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/biobakery/humann/blob/e07b3a34d0b94c09a8ac5d28ff95009611178be2/readme.md","artifact_url":"https://raw.githubusercontent.com/biobakery/humann/e07b3a34d0b94c09a8ac5d28ff95009611178be2/readme.md","version":"e07b3a34d0b94c09a8ac5d28ff95009611178be2","retrieved_at":"2026-09-16T19:46:18.580457+00:00","artifact_sha256":"96260519b594ae22cba9f28d1f64622de001f2abf11d406c9da572bfaf145727","locator":"readme.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-116050b81d03a38e3b6f","kind":"source","name":"dreams: Journal full-text XML","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13090125/","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13090125/fullTextXML","version":"Retrieved page snapshot; no immutable publisher revision supplied","retrieved_at":"2026-09-16T19:53:03.200000+00:00","artifact_sha256":"4bbe2ecff0ad75944b3aec5131369c1b2f75ff6f4b5dc257487a274fdd7b0fef","locator":"Journal full-text XML","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-116dfd8f61e04b51c148","kind":"source","name":"tbepler/protein-sequence-embedding-iclr2019: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/tbepler/protein-sequence-embedding-iclr2019/blob/be32cffeec26431bdf87438eb5f07ddd6fc5d7dd/README.md","artifact_url":"https://raw.githubusercontent.com/tbepler/protein-sequence-embedding-iclr2019/be32cffeec26431bdf87438eb5f07ddd6fc5d7dd/README.md","version":"be32cffeec26431bdf87438eb5f07ddd6fc5d7dd","retrieved_at":"2026-09-16T19:57:33.688522+00:00","artifact_sha256":"ecb193485e89d65558a509bae571057862dd079d29afc877c8c9812c4c525415","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-12d43bf3076a93fc4ee7","kind":"source","name":"songlab-cal/tape: tape/models/modeling_onehot.py","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/songlab-cal/tape/blob/6d345c2b2bbf52cd32cf179325c222afd92aec7e/tape/models/modeling_onehot.py","artifact_url":"https://raw.githubusercontent.com/songlab-cal/tape/6d345c2b2bbf52cd32cf179325c222afd92aec7e/tape/models/modeling_onehot.py","version":"6d345c2b2bbf52cd32cf179325c222afd92aec7e","retrieved_at":"2026-09-16T19:46:20.570146+00:00","artifact_sha256":"546fab3931d2ae7b6a0ee06ff64c6bcedf2d58b460e5aa90be623b0895befe43","locator":"tape/models/modeling_onehot.py","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-137a3a936ffa4673b3d3","kind":"source","name":"facebook/esm1v_t33_650M_UR90S_1: config.json","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://huggingface.co/facebook/esm1v_t33_650M_UR90S_1/blob/8bfdb1892536cc77bd0760b9c25ddced2cd0b4c8/config.json","artifact_url":"https://huggingface.co/facebook/esm1v_t33_650M_UR90S_1/blob/8bfdb1892536cc77bd0760b9c25ddced2cd0b4c8/config.json","version":"8bfdb1892536cc77bd0760b9c25ddced2cd0b4c8","retrieved_at":"2026-09-16T20:12:20.912868+00:00","artifact_sha256":"6c4b576e2e73ad85fc51a4fa75d4e350532b773e04551307a74643c3fc3c5d59","locator":"config.json","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-18d8f4d5fc3f6616922d","kind":"source","name":"instadeepai/nucleotide-transformer: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/instadeepai/nucleotide-transformer/blob/2dc37b86e16a6970fbc731751f7719d9f676f7f9/README.md","artifact_url":"https://raw.githubusercontent.com/instadeepai/nucleotide-transformer/2dc37b86e16a6970fbc731751f7719d9f676f7f9/README.md","version":"2dc37b86e16a6970fbc731751f7719d9f676f7f9","retrieved_at":"2026-09-16T19:46:19.364532+00:00","artifact_sha256":"9f51bbb20c4c5c36e77fb03ca1c5c36236e287c48a1ee31f53150545d421ec25","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-1a0775cac85be75449ef","kind":"source","name":"biobakery/MetaPhlAn: license.txt","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/biobakery/MetaPhlAn/blob/424f3e6e30618266404353e1083c6405a9f02f48/license.txt","artifact_url":"https://raw.githubusercontent.com/biobakery/MetaPhlAn/424f3e6e30618266404353e1083c6405a9f02f48/license.txt","version":"424f3e6e30618266404353e1083c6405a9f02f48","retrieved_at":"2026-09-16T19:46:18.561509+00:00","artifact_sha256":"ecf18c2928e49997ba1f098f0f0da0958d257d69fffb48875162dbd24f3d7762","locator":"license.txt","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-1aae5efd1b27ae9c761a","kind":"source","name":"songlab-cal/tape: tape/models/modeling_lstm.py","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/songlab-cal/tape/blob/6d345c2b2bbf52cd32cf179325c222afd92aec7e/tape/models/modeling_lstm.py","artifact_url":"https://raw.githubusercontent.com/songlab-cal/tape/6d345c2b2bbf52cd32cf179325c222afd92aec7e/tape/models/modeling_lstm.py","version":"6d345c2b2bbf52cd32cf179325c222afd92aec7e","retrieved_at":"2026-09-16T19:46:20.570146+00:00","artifact_sha256":"e857bc4d6d551915a375116208f7f74b286e7fc88d0d54f54b2746b2fb02e1ca","locator":"tape/models/modeling_lstm.py","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-1b2aa7207e3a05f297ab","kind":"source","name":"ViennaRNA/ViennaRNA: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/ViennaRNA/ViennaRNA/blob/1ffec79f5e258896160f7362ced8263450f371dc/README.md","artifact_url":"https://raw.githubusercontent.com/ViennaRNA/ViennaRNA/1ffec79f5e258896160f7362ced8263450f371dc/README.md","version":"1ffec79f5e258896160f7362ced8263450f371dc","retrieved_at":"2026-09-16T19:46:18.166113+00:00","artifact_sha256":"d37146b01e4273062a5230c496a9af8414ee7ef14fcd905cf415959256d59c4e","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-1d81a48c6dcc0e6dc14d","kind":"source","name":"biohub/ESMC-6B: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://huggingface.co/biohub/ESMC-6B/blob/af1602ba7406f521b11bf8f81d52af378cde09e4/README.md","artifact_url":"https://huggingface.co/biohub/ESMC-6B/blob/af1602ba7406f521b11bf8f81d52af378cde09e4/README.md","version":"af1602ba7406f521b11bf8f81d52af378cde09e4","retrieved_at":"2026-09-16T20:22:27.767420+00:00","artifact_sha256":"64096baa2f7345c1badd5be2c97c0998fc46d4c3d017c2a3f3ce83418a20f147","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-1df5b1865861177c0c75","kind":"source","name":"ArcInstitute/state: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/ArcInstitute/state/blob/9bbfe78a434a55205e4de834e1ea99f85f7a3add/README.md","artifact_url":"https://raw.githubusercontent.com/ArcInstitute/state/9bbfe78a434a55205e4de834e1ea99f85f7a3add/README.md","version":"9bbfe78a434a55205e4de834e1ea99f85f7a3add","retrieved_at":"2026-09-16T19:46:17.767683+00:00","artifact_sha256":"568c0b4f9d93374f7ebc7467c226fdde0050abfd60147bce0b316163bf4199d6","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-1f027f8ecc53decdeec8","kind":"source","name":"evolutionaryscale/esm: THIRD_PARTY_NOTICE.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/evolutionaryscale/esm/blob/bf343ba264b650dff7a073643725f9aaa1fdbe8d/THIRD_PARTY_NOTICE.md","artifact_url":"https://raw.githubusercontent.com/evolutionaryscale/esm/bf343ba264b650dff7a073643725f9aaa1fdbe8d/THIRD_PARTY_NOTICE.md","version":"bf343ba264b650dff7a073643725f9aaa1fdbe8d","retrieved_at":"2026-09-16T19:46:18.983347+00:00","artifact_sha256":"5bff8515ba4e0f53abdc43714c180b79c5b606160497d98de741a369cb9b6a23","locator":"THIRD_PARTY_NOTICE.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-1f756fefa823dd402b0f","kind":"source","name":"kraken2: Journal full-text XML","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC6883579/","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC6883579/fullTextXML","version":"Retrieved page snapshot; no immutable publisher revision supplied","retrieved_at":"2026-09-16T19:53:03.009224+00:00","artifact_sha256":"2fd1d7785b45855deacb566acc79336e292c4c2d4975095b2ff7b1200c406869","locator":"Journal full-text XML","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-23e72f7e93a0a633ab2b","kind":"source","name":"ccsb-scripps/AutoDock-Vina: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/ccsb-scripps/AutoDock-Vina/blob/3c65c0b3e6c2c1d183f6a175ecb65e3c5ba91645/README.md","artifact_url":"https://raw.githubusercontent.com/ccsb-scripps/AutoDock-Vina/3c65c0b3e6c2c1d183f6a175ecb65e3c5ba91645/README.md","version":"3c65c0b3e6c2c1d183f6a175ecb65e3c5ba91645","retrieved_at":"2026-09-16T19:46:18.718558+00:00","artifact_sha256":"4f1728521ab79de1c33e1cf8605b31037effed5de2a2fbbccba58d7b0a005ae7","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-25b222d11900e0e88a51","kind":"source","name":"MAGICS-LAB/DNABERT_2: LICENSE","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/MAGICS-LAB/DNABERT_2/blob/f25bed9ee20db966dff39e5c1571249d04e36404/LICENSE","artifact_url":"https://raw.githubusercontent.com/MAGICS-LAB/DNABERT_2/f25bed9ee20db966dff39e5c1571249d04e36404/LICENSE","version":"f25bed9ee20db966dff39e5c1571249d04e36404","retrieved_at":"2026-09-16T19:46:17.892989+00:00","artifact_sha256":"c71d239df91726fc519c6eb72d318ec65820627232b2f796219e87dcf35d0ab4","locator":"LICENSE","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-28c1720afecf4240f3f7","kind":"source","name":"rfdiffusion-supp: Publisher supplementary archive","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC10468394/supplementaryFiles","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC10468394/supplementaryFiles","version":"Retrieved page snapshot; no immutable publisher revision supplied","retrieved_at":"2026-09-16T20:43:11.978663+00:00","artifact_sha256":"eb2ee3da27da262aa90fd5c4ee371581e429039b2b976c135463fbe4b2dbb199","locator":"Publisher supplementary archive","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-28fc17fbe0c1d99da219","kind":"source","name":"facebook/esm2_t33_650M_UR50D: config.json","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://huggingface.co/facebook/esm2_t33_650M_UR50D/blob/08e4846e537177426273712802403f7ba8261b6c/config.json","artifact_url":"https://huggingface.co/facebook/esm2_t33_650M_UR50D/blob/08e4846e537177426273712802403f7ba8261b6c/config.json","version":"08e4846e537177426273712802403f7ba8261b6c","retrieved_at":"2026-09-16T20:04:02.230771+00:00","artifact_sha256":"539095c22efc52a09d6147074ba4ca119f76a890df5901213b2b55f7d2f96b2b","locator":"config.json","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-290b60bb932abe9929a2","kind":"source","name":"chaidiscovery/chai-lab: LICENSE","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/chaidiscovery/chai-lab/blob/66c38d1fe5c6756a89ff8596b1dea87d305ec06f/LICENSE","artifact_url":"https://raw.githubusercontent.com/chaidiscovery/chai-lab/66c38d1fe5c6756a89ff8596b1dea87d305ec06f/LICENSE","version":"66c38d1fe5c6756a89ff8596b1dea87d305ec06f","retrieved_at":"2026-09-16T19:46:18.824295+00:00","artifact_sha256":"511edf51c5c6f47bae9ae19c59d98666a682e0ccc98e90c2de2a9a897c44c003","locator":"LICENSE","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-29d4b229a3aa426a6dfb","kind":"source","name":"facebookresearch/esm: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/facebookresearch/esm/blob/2b369911bb5b4b0dda914521b9475cad1656b2ac/README.md","artifact_url":"https://raw.githubusercontent.com/facebookresearch/esm/2b369911bb5b4b0dda914521b9475cad1656b2ac/README.md","version":"2b369911bb5b4b0dda914521b9475cad1656b2ac","retrieved_at":"2026-09-16T19:46:19.090082+00:00","artifact_sha256":"8b273c21a322fc9473d1b68d0dd40c8166ab2f89e4a190aa26ca87251b97cba9","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-2bf115f7d072744af00a","kind":"source","name":"BojarLab/glycowork: glycowork/ml/models.py","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/BojarLab/glycowork/blob/3d63f1ec25c850da3cde4d25cb602d50b6b5732b/glycowork/ml/models.py","artifact_url":"https://raw.githubusercontent.com/BojarLab/glycowork/3d63f1ec25c850da3cde4d25cb602d50b6b5732b/glycowork/ml/models.py","version":"3d63f1ec25c850da3cde4d25cb602d50b6b5732b","retrieved_at":"2026-09-16T19:46:17.767851+00:00","artifact_sha256":"40222af745e6ec5795ff03d86ca33f9d40eaf331330cd677d9013fce01e88bfb","locator":"glycowork/ml/models.py","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-2cf0a41fec83ee9c5cc9","kind":"source","name":"zhihan1996/DNABERT-2-117M: config.json","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://huggingface.co/zhihan1996/DNABERT-2-117M/blob/7bce263b15377fc15361f52cfab88f8b586abda0/config.json","artifact_url":"https://huggingface.co/zhihan1996/DNABERT-2-117M/blob/7bce263b15377fc15361f52cfab88f8b586abda0/config.json","version":"7bce263b15377fc15361f52cfab88f8b586abda0","retrieved_at":"2026-09-16T19:46:20.874956+00:00","artifact_sha256":"ba9bdafaff0cc3e30556927474d4a179519a9864012bed2628e9f1bc23c84bfd","locator":"config.json","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-2e7c7649620407f50f6b","kind":"source","name":"facebookresearch/esm: LICENSE","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/facebookresearch/esm/blob/2b369911bb5b4b0dda914521b9475cad1656b2ac/LICENSE","artifact_url":"https://raw.githubusercontent.com/facebookresearch/esm/2b369911bb5b4b0dda914521b9475cad1656b2ac/LICENSE","version":"2b369911bb5b4b0dda914521b9475cad1656b2ac","retrieved_at":"2026-09-16T19:46:19.090082+00:00","artifact_sha256":"da6d3703ed11cbe42bd212c725957c98da23cbff1998c05fa4b3d976d1a58e93","locator":"LICENSE","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-2efbfaba5a1c09f4f7fa","kind":"source","name":"nbrg-ppcu/prokbert: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/nbrg-ppcu/prokbert/blob/8670ae92b816cff158a0b85647a8dea122e251eb/README.md","artifact_url":"https://raw.githubusercontent.com/nbrg-ppcu/prokbert/8670ae92b816cff158a0b85647a8dea122e251eb/README.md","version":"8670ae92b816cff158a0b85647a8dea122e251eb","retrieved_at":"2026-09-16T19:46:19.913949+00:00","artifact_sha256":"29de39c6ad006ce704ab14240cfd97af93da411fb63ec89ebe499c2646928cfc","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-2f4711ecd64b0162e85b","kind":"source","name":"evolutionaryscale/esm: LICENSE.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/evolutionaryscale/esm/blob/bf343ba264b650dff7a073643725f9aaa1fdbe8d/LICENSE.md","artifact_url":"https://raw.githubusercontent.com/evolutionaryscale/esm/bf343ba264b650dff7a073643725f9aaa1fdbe8d/LICENSE.md","version":"bf343ba264b650dff7a073643725f9aaa1fdbe8d","retrieved_at":"2026-09-16T19:46:18.983347+00:00","artifact_sha256":"b63df9ca1dd96b3b21eec226b51b236d0bd152ac20eafc43aad46bf832b48d8a","locator":"LICENSE.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-3076c52b2fa7cb48ba7b","kind":"source","name":"https://zenodo.org/api/records/10997887: page.html","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://zenodo.org/api/records/10997887","artifact_url":"https://zenodo.org/api/records/10997887","version":"Retrieved page snapshot; no immutable publisher revision supplied","retrieved_at":"2026-09-16T20:22:27.767327+00:00","artifact_sha256":"bfd7a787f3a7f1f900ca62b763a7161a486653011a43b50c39911a3b82d46c9e","locator":"page.html","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-33fe0374c8ebbc2c7294","kind":"source","name":"DerrickWood/kraken2: docs/MANUAL.markdown","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/DerrickWood/kraken2/blob/8c190b1b668825935dbf6dee5f969227dc8269bb/docs/MANUAL.markdown","artifact_url":"https://raw.githubusercontent.com/DerrickWood/kraken2/8c190b1b668825935dbf6dee5f969227dc8269bb/docs/MANUAL.markdown","version":"8c190b1b668825935dbf6dee5f969227dc8269bb","retrieved_at":"2026-09-16T19:46:17.767980+00:00","artifact_sha256":"182565cb02f3958b39b8303749e58cd8829c6c6bf4734e39533d86e7980071f7","locator":"docs/MANUAL.markdown","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-34faacf1ad53d6aa6bef","kind":"source","name":"PolymathicAI/MIMIC: LICENSE","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/PolymathicAI/MIMIC/blob/9e652f16491e6c2e3881111e24b285c288554275/LICENSE","artifact_url":"https://raw.githubusercontent.com/PolymathicAI/MIMIC/9e652f16491e6c2e3881111e24b285c288554275/LICENSE","version":"9e652f16491e6c2e3881111e24b285c288554275","retrieved_at":"2026-09-16T20:04:02.231308+00:00","artifact_sha256":"cb4951b3d04c3153a6192951a520d7042c927d1f0574978d503596d6c14e36c9","locator":"LICENSE","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-362b9cef071111dfe343","kind":"source","name":"tape: Primary paper PDF","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://arxiv.org/pdf/1906.08230","artifact_url":"https://arxiv.org/pdf/1906.08230","version":"1906.08230v1","retrieved_at":"2026-09-16T20:23:48.238777+00:00","artifact_sha256":"6ee0c3e6e870635cba8fa67e0a4abc5598c0ab2a10ba127a46b67ac450ae0168","locator":"Primary paper PDF","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-3759a2f34cb49263eb43","kind":"source","name":"ODonnell-Lipidomics/LipidFinder: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/ODonnell-Lipidomics/LipidFinder/blob/8306ca014c2e6b34ce6ef3ec6cab01fa4c666a09/README.md","artifact_url":"https://raw.githubusercontent.com/ODonnell-Lipidomics/LipidFinder/8306ca014c2e6b34ce6ef3ec6cab01fa4c666a09/README.md","version":"8306ca014c2e6b34ce6ef3ec6cab01fa4c666a09","retrieved_at":"2026-09-16T19:57:35.576108+00:00","artifact_sha256":"0449bf076d0d37a2690f0e66f655a03d5e3577207e166f2c963d93d3afcbcd1a","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-37f825d1226e59cbbfbf","kind":"source","name":"metaphlan: Journal full-text XML","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC10635831/","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC10635831/fullTextXML","version":"Retrieved page snapshot; no immutable publisher revision supplied","retrieved_at":"2026-09-16T19:53:03.009254+00:00","artifact_sha256":"547306680d52f70865906db567dee93323ecdec3b2e162608dff5057e057aa4b","locator":"Journal full-text XML","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-39a15ed8072aeea14a22","kind":"source","name":"jwohlwend/boltz: docs/training.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/jwohlwend/boltz/blob/b1ebfc46ecf57f5414e0d1a6f9027bbb122c53bc/docs/training.md","artifact_url":"https://raw.githubusercontent.com/jwohlwend/boltz/b1ebfc46ecf57f5414e0d1a6f9027bbb122c53bc/docs/training.md","version":"b1ebfc46ecf57f5414e0d1a6f9027bbb122c53bc","retrieved_at":"2026-09-16T19:46:19.465940+00:00","artifact_sha256":"4574fa3cac9d086708dce59de68a144dc337722af3242db9adfd6798bb37e21c","locator":"docs/training.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-3aa749d3205c0a768b22","kind":"source","name":"dnabert2: Primary paper PDF","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://arxiv.org/pdf/2306.15006","artifact_url":"https://arxiv.org/pdf/2306.15006","version":"2306.15006v2","retrieved_at":"2026-09-16T20:04:04.777189+00:00","artifact_sha256":"49300acee3e4afd44bebc3de9893c3bc310d331bd4805374e0952fdfbf366f06","locator":"Primary paper PDF","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-3cec54ed6ded9cb0a092","kind":"source","name":"songlab-cal/tape: tape/models/modeling_resnet.py","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/songlab-cal/tape/blob/6d345c2b2bbf52cd32cf179325c222afd92aec7e/tape/models/modeling_resnet.py","artifact_url":"https://raw.githubusercontent.com/songlab-cal/tape/6d345c2b2bbf52cd32cf179325c222afd92aec7e/tape/models/modeling_resnet.py","version":"6d345c2b2bbf52cd32cf179325c222afd92aec7e","retrieved_at":"2026-09-16T19:46:20.570146+00:00","artifact_sha256":"fc97f083f059f5b587d568aa0c32f7cca227e247160a7e22e3a168aef1469ebe","locator":"tape/models/modeling_resnet.py","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-3e1890c3652bc5c150f7","kind":"source","name":"rhofold: Journal full-text XML","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11621015/","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11621015/fullTextXML","version":"Retrieved page snapshot; no immutable publisher revision supplied","retrieved_at":"2026-09-16T19:53:03.009162+00:00","artifact_sha256":"a74ba0e47c4b0cdc4481f10ffd10d323eaabc9906d55315f9b2009898d8ae803","locator":"Journal full-text XML","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-3fde3df73e79e455bd86","kind":"source","name":"ml4bio/RNA-FM: LICENSE","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/ml4bio/RNA-FM/blob/348951516e0963d22bbb33b3c9fc18c89081d38e/LICENSE","artifact_url":"https://raw.githubusercontent.com/ml4bio/RNA-FM/348951516e0963d22bbb33b3c9fc18c89081d38e/LICENSE","version":"348951516e0963d22bbb33b3c9fc18c89081d38e","retrieved_at":"2026-09-16T19:46:19.769679+00:00","artifact_sha256":"b0809e99b532fdf51660f5a3d2a9010ed09d15aef0d131ad80fe80c2291a4fba","locator":"LICENSE","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-40f46dc4b9fcc306773b","kind":"source","name":"DerrickWood/kraken2: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/DerrickWood/kraken2/blob/8c190b1b668825935dbf6dee5f969227dc8269bb/README.md","artifact_url":"https://raw.githubusercontent.com/DerrickWood/kraken2/8c190b1b668825935dbf6dee5f969227dc8269bb/README.md","version":"8c190b1b668825935dbf6dee5f969227dc8269bb","retrieved_at":"2026-09-16T19:46:17.767980+00:00","artifact_sha256":"2ea33af266b4268a55fd750d0f3265cd3165d61f6be375c5ea3b3ff5c58c7c8c","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-4269946a73db94a15289","kind":"source","name":"ViennaRNA/ViennaRNA: COPYING","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/ViennaRNA/ViennaRNA/blob/1ffec79f5e258896160f7362ced8263450f371dc/COPYING","artifact_url":"https://raw.githubusercontent.com/ViennaRNA/ViennaRNA/1ffec79f5e258896160f7362ced8263450f371dc/COPYING","version":"1ffec79f5e258896160f7362ced8263450f371dc","retrieved_at":"2026-09-16T19:46:18.166113+00:00","artifact_sha256":"7776cce4c6155eba82ea6bcb5c108f25b9fb0f604664421bfd6da94930028d8c","locator":"COPYING","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-43484d4de29aacd65ed7","kind":"source","name":"bowang-lab/scGPT: LICENSE","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/bowang-lab/scGPT/blob/cebd6fae655b9c585a4807daa3ac31bb764f06b4/LICENSE","artifact_url":"https://raw.githubusercontent.com/bowang-lab/scGPT/cebd6fae655b9c585a4807daa3ac31bb764f06b4/LICENSE","version":"cebd6fae655b9c585a4807daa3ac31bb764f06b4","retrieved_at":"2026-09-16T19:46:18.594429+00:00","artifact_sha256":"1ceeacbed51e2890187425547bc2efd16c1b7ad45189b7dfb21e83a45a2e9d9e","locator":"LICENSE","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-4565e66fdeb787b99afa","kind":"source","name":"polymathic-ai/MIMIC: config.json","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://huggingface.co/polymathic-ai/MIMIC/blob/72e63a1ece34928422fd46f89b6a6580ace99a97/config.json","artifact_url":"https://huggingface.co/polymathic-ai/MIMIC/blob/72e63a1ece34928422fd46f89b6a6580ace99a97/config.json","version":"72e63a1ece34928422fd46f89b6a6580ace99a97","retrieved_at":"2026-09-16T19:46:20.846683+00:00","artifact_sha256":"687e182e410fbfa235e24c337f51411c6fff583d3135fdbac202b08b0e63a8ef","locator":"config.json","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-45f05eebcf7a03ac99cf","kind":"source","name":"kundajelab/chrombpnet: LICENSE","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/kundajelab/chrombpnet/blob/09938fdb4397ec0006510e5251e48920a505d4de/LICENSE","artifact_url":"https://raw.githubusercontent.com/kundajelab/chrombpnet/09938fdb4397ec0006510e5251e48920a505d4de/LICENSE","version":"09938fdb4397ec0006510e5251e48920a505d4de","retrieved_at":"2026-09-16T19:46:19.514840+00:00","artifact_sha256":"eee7b4d55be619630ce91024410485d25226c387e72315b83712a8dbf89188ef","locator":"LICENSE","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-4603e2d255507e025c80","kind":"source","name":"InstaDeepAI/agro-nucleotide-transformer-1b: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://huggingface.co/InstaDeepAI/agro-nucleotide-transformer-1b/blob/b0e1ea1f53a2bf5bb29f8eab7a7e553bf06c1ab1/README.md","artifact_url":"https://huggingface.co/InstaDeepAI/agro-nucleotide-transformer-1b/blob/b0e1ea1f53a2bf5bb29f8eab7a7e553bf06c1ab1/README.md","version":"b0e1ea1f53a2bf5bb29f8eab7a7e553bf06c1ab1","retrieved_at":"2026-09-16T20:12:20.913125+00:00","artifact_sha256":"552b83763443130a7a207a749d47026c5cc36be72268cb827afe251c30857807","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-4920952f9b3c4b8909a0","kind":"source","name":"instadeepai/nucleotide-transformer: docs/segment_nt.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/instadeepai/nucleotide-transformer/blob/2dc37b86e16a6970fbc731751f7719d9f676f7f9/docs/segment_nt.md","artifact_url":"https://raw.githubusercontent.com/instadeepai/nucleotide-transformer/2dc37b86e16a6970fbc731751f7719d9f676f7f9/docs/segment_nt.md","version":"2dc37b86e16a6970fbc731751f7719d9f676f7f9","retrieved_at":"2026-09-16T19:46:19.364532+00:00","artifact_sha256":"8eec4580ba64ab944fb9b42674be70ffe793135f3503cce8640f5b08f8290f7a","locator":"docs/segment_nt.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-4abb9affb1a9e5438c91","kind":"source","name":"InstaDeepAI/nucleotide-transformer-v2-50m-multi-species: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://huggingface.co/InstaDeepAI/nucleotide-transformer-v2-50m-multi-species/blob/81b29e5786726d891dbf929404ef20adca5b36f1/README.md","artifact_url":"https://huggingface.co/InstaDeepAI/nucleotide-transformer-v2-50m-multi-species/blob/81b29e5786726d891dbf929404ef20adca5b36f1/README.md","version":"81b29e5786726d891dbf929404ef20adca5b36f1","retrieved_at":"2026-09-16T19:46:20.607357+00:00","artifact_sha256":"e526d7b98f106bc2ca9ba73fa166ff5fd62853812e9757a6692eeedf25e42923","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-4bf005200ebfb2586b78","kind":"source","name":"evolutionaryscale/esm: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/evolutionaryscale/esm/blob/bf343ba264b650dff7a073643725f9aaa1fdbe8d/README.md","artifact_url":"https://raw.githubusercontent.com/evolutionaryscale/esm/bf343ba264b650dff7a073643725f9aaa1fdbe8d/README.md","version":"bf343ba264b650dff7a073643725f9aaa1fdbe8d","retrieved_at":"2026-09-16T19:46:18.983347+00:00","artifact_sha256":"74a897f0e97e3d4256f7cff11424dd264df4daaf8cabd9a6fb74b661421de038","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-4e0280294c1215534077","kind":"source","name":"ml4bio/RhoFold: LICENSE","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/ml4bio/RhoFold/blob/6bdfbda720184409eb682ce08c05d258162ddc48/LICENSE","artifact_url":"https://raw.githubusercontent.com/ml4bio/RhoFold/6bdfbda720184409eb682ce08c05d258162ddc48/LICENSE","version":"6bdfbda720184409eb682ce08c05d258162ddc48","retrieved_at":"2026-09-16T19:46:19.891935+00:00","artifact_sha256":"cfc7749b96f63bd31c3c42b5c471bf756814053e847c10f3eb003417bc523d30","locator":"LICENSE","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-53934b67953ba33b2c95","kind":"source","name":"aqlaboratory/openfold: docs/source/original_readme.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/aqlaboratory/openfold/blob/be2ec1841f16c966c65ae0e7599ebbadc725757d/docs/source/original_readme.md","artifact_url":"https://raw.githubusercontent.com/aqlaboratory/openfold/be2ec1841f16c966c65ae0e7599ebbadc725757d/docs/source/original_readme.md","version":"be2ec1841f16c966c65ae0e7599ebbadc725757d","retrieved_at":"2026-09-16T19:46:18.314650+00:00","artifact_sha256":"aa0812608f4b4dca4b4d30b9e0eb6b7d08ca9d1596535419d6255c3ced48f36f","locator":"docs/source/original_readme.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-550f6719c2f22f4d57b3","kind":"source","name":"scverse/scvi-tools: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/scverse/scvi-tools/blob/73b28e44223621470e582a81a102c107bb22678b/README.md","artifact_url":"https://raw.githubusercontent.com/scverse/scvi-tools/73b28e44223621470e582a81a102c107bb22678b/README.md","version":"73b28e44223621470e582a81a102c107bb22678b","retrieved_at":"2026-09-16T19:46:20.137176+00:00","artifact_sha256":"eb46b8a54e60643ca0cd8cb375ede05b01dcbd2380ca17ce8027a92ba13cebbb","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-5631fb75aa8e5a7a89e1","kind":"source","name":"ODonnell-Lipidomics/LipidFinder: LICENSE","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/ODonnell-Lipidomics/LipidFinder/blob/8306ca014c2e6b34ce6ef3ec6cab01fa4c666a09/LICENSE","artifact_url":"https://raw.githubusercontent.com/ODonnell-Lipidomics/LipidFinder/8306ca014c2e6b34ce6ef3ec6cab01fa4c666a09/LICENSE","version":"8306ca014c2e6b34ce6ef3ec6cab01fa4c666a09","retrieved_at":"2026-09-16T19:57:35.576108+00:00","artifact_sha256":"86608efc82f1b2b37ff0ae607d15cd4cd5f779b9a1dd8fe0410c7634216bf9be","locator":"LICENSE","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-56e5abfb5f12f1cd3b20","kind":"source","name":"alphagenome: Journal full-text XML","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12851941/","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12851941/fullTextXML","version":"Retrieved page snapshot; no immutable publisher revision supplied","retrieved_at":"2026-09-16T19:53:03.009088+00:00","artifact_sha256":"d159b791fc6cb7b679c151727b7b05a6e5b6388b08a89b675014983d85b6df36","locator":"Journal full-text XML","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-56f02d45976d011d80aa","kind":"source","name":"instadeepai/nucleotide-transformer: docs/agro_nucleotide_transformer.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/instadeepai/nucleotide-transformer/blob/2dc37b86e16a6970fbc731751f7719d9f676f7f9/docs/agro_nucleotide_transformer.md","artifact_url":"https://raw.githubusercontent.com/instadeepai/nucleotide-transformer/2dc37b86e16a6970fbc731751f7719d9f676f7f9/docs/agro_nucleotide_transformer.md","version":"2dc37b86e16a6970fbc731751f7719d9f676f7f9","retrieved_at":"2026-09-16T19:46:19.364532+00:00","artifact_sha256":"23e27d40473bbabadac45c56e8e282349f0b053da85fda89ff2125b5fe381fc6","locator":"docs/agro_nucleotide_transformer.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-576c2ecba240ddfda8f2","kind":"source","name":"agront: Journal full-text XML","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11233511/","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11233511/fullTextXML","version":"Retrieved page snapshot; no immutable publisher revision supplied","retrieved_at":"2026-09-16T20:16:14.422658+00:00","artifact_sha256":"ac36567140994e07e029042e4ac088d5895eabc2cf7ee79b174aee7d7a074070","locator":"Journal full-text XML","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-578439da3f3a3a476b7f","kind":"source","name":"google-deepmind/alphagenome_research: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/google-deepmind/alphagenome_research/blob/0db53bd4352c66d1e00a049a81da373a066e6670/README.md","artifact_url":"https://raw.githubusercontent.com/google-deepmind/alphagenome_research/0db53bd4352c66d1e00a049a81da373a066e6670/README.md","version":"0db53bd4352c66d1e00a049a81da373a066e6670","retrieved_at":"2026-09-16T19:46:19.361780+00:00","artifact_sha256":"c715b22c31bd4f69e9b76236de0ab6a26c96fa0f384ad56a2f9d2cb65e262681","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-5a9e55abdf288d9dd8da","kind":"source","name":"jwohlwend/boltz: docs/prediction.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/jwohlwend/boltz/blob/b1ebfc46ecf57f5414e0d1a6f9027bbb122c53bc/docs/prediction.md","artifact_url":"https://raw.githubusercontent.com/jwohlwend/boltz/b1ebfc46ecf57f5414e0d1a6f9027bbb122c53bc/docs/prediction.md","version":"b1ebfc46ecf57f5414e0d1a6f9027bbb122c53bc","retrieved_at":"2026-09-16T19:46:19.465940+00:00","artifact_sha256":"b9cb2ff437389864bde02e7e9fd9fbcc7de189ca2e240a5f3fb6261ca595a795","locator":"docs/prediction.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-5e268f31f0347fc5564c","kind":"source","name":"prokbert: Journal full-text XML","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC10810988/","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC10810988/fullTextXML","version":"Retrieved page snapshot; no immutable publisher revision supplied","retrieved_at":"2026-09-16T19:53:03.847832+00:00","artifact_sha256":"8610e2a54aa877c8dc565a9cdb6e82099f284c5e0907a52cab18d994ea732436","locator":"Journal full-text XML","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-6189052c702a948da02d","kind":"source","name":"biomap-research/scFoundation: LICENSE","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/biomap-research/scFoundation/blob/397631c495eddf9ad6644fc00c6ea8139e651245/LICENSE","artifact_url":"https://raw.githubusercontent.com/biomap-research/scFoundation/397631c495eddf9ad6644fc00c6ea8139e651245/LICENSE","version":"397631c495eddf9ad6644fc00c6ea8139e651245","retrieved_at":"2026-09-16T19:46:18.581063+00:00","artifact_sha256":"c71d239df91726fc519c6eb72d318ec65820627232b2f796219e87dcf35d0ab4","locator":"LICENSE","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-640665f30e62eed9319b","kind":"source","name":"ccsb-scripps/AutoDock-Vina: LICENSE","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/ccsb-scripps/AutoDock-Vina/blob/3c65c0b3e6c2c1d183f6a175ecb65e3c5ba91645/LICENSE","artifact_url":"https://raw.githubusercontent.com/ccsb-scripps/AutoDock-Vina/3c65c0b3e6c2c1d183f6a175ecb65e3c5ba91645/LICENSE","version":"3c65c0b3e6c2c1d183f6a175ecb65e3c5ba91645","retrieved_at":"2026-09-16T19:46:18.718558+00:00","artifact_sha256":"cfc7749b96f63bd31c3c42b5c471bf756814053e847c10f3eb003417bc523d30","locator":"LICENSE","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-657e83427ab59f3aec83","kind":"source","name":"instadeepai/nucleotide-transformer: docs/nucleotide_transformer.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/instadeepai/nucleotide-transformer/blob/2dc37b86e16a6970fbc731751f7719d9f676f7f9/docs/nucleotide_transformer.md","artifact_url":"https://raw.githubusercontent.com/instadeepai/nucleotide-transformer/2dc37b86e16a6970fbc731751f7719d9f676f7f9/docs/nucleotide_transformer.md","version":"2dc37b86e16a6970fbc731751f7719d9f676f7f9","retrieved_at":"2026-09-16T19:46:19.364532+00:00","artifact_sha256":"ab16d582de98652526b5cebb120eec969328f9db29dc741826bcd81c397e0672","locator":"docs/nucleotide_transformer.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-67ee1cc31060ba8c9569","kind":"source","name":"ArcInstitute/evo2: LICENSE","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/ArcInstitute/evo2/blob/53f195997257c56c00e5ef8d33a54f5baad143a6/LICENSE","artifact_url":"https://raw.githubusercontent.com/ArcInstitute/evo2/53f195997257c56c00e5ef8d33a54f5baad143a6/LICENSE","version":"53f195997257c56c00e5ef8d33a54f5baad143a6","retrieved_at":"2026-09-16T19:46:17.765915+00:00","artifact_sha256":"5bb5812fc2bfb2d777fe5621767172f8ef30be62ad719c87e0f10832336a99e0","locator":"LICENSE","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-6851724e3bcb7e9d2781","kind":"source","name":"scverse/scvi-tools: LICENSE","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/scverse/scvi-tools/blob/73b28e44223621470e582a81a102c107bb22678b/LICENSE","artifact_url":"https://raw.githubusercontent.com/scverse/scvi-tools/73b28e44223621470e582a81a102c107bb22678b/LICENSE","version":"73b28e44223621470e582a81a102c107bb22678b","retrieved_at":"2026-09-16T19:46:20.137176+00:00","artifact_sha256":"66399db0284d2539790efb348886ab0c1f745bbe5fea6ac38a00465a14adc8f5","locator":"LICENSE","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-6a331ce74fa7d11e0bfe","kind":"source","name":"biohub/ESMFold2: config.json","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://huggingface.co/biohub/ESMFold2/blob/69869f737beffec5294845ede23db5fc0b4f509e/config.json","artifact_url":"https://huggingface.co/biohub/ESMFold2/blob/69869f737beffec5294845ede23db5fc0b4f509e/config.json","version":"69869f737beffec5294845ede23db5fc0b4f509e","retrieved_at":"2026-09-16T20:22:27.767500+00:00","artifact_sha256":"72054c1af92f432ac2d8e0628ba08b377d04c6a8679b5128b798dbc7f644ddc8","locator":"config.json","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-6b79ddfdd330693bf4fb","kind":"source","name":"InstaDeepAI/nucleotide-transformer-v2-50m-multi-species: config.json","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://huggingface.co/InstaDeepAI/nucleotide-transformer-v2-50m-multi-species/blob/81b29e5786726d891dbf929404ef20adca5b36f1/config.json","artifact_url":"https://huggingface.co/InstaDeepAI/nucleotide-transformer-v2-50m-multi-species/blob/81b29e5786726d891dbf929404ef20adca5b36f1/config.json","version":"81b29e5786726d891dbf929404ef20adca5b36f1","retrieved_at":"2026-09-16T19:46:20.607357+00:00","artifact_sha256":"e20f497248c7cb264c7cd4582dbcfd52dc4cbf74a97fc711559b8c8f71c635db","locator":"config.json","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-6bcab1b3e31b52c38ea5","kind":"source","name":"BojarLab/glycowork: LICENSE","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/BojarLab/glycowork/blob/3d63f1ec25c850da3cde4d25cb602d50b6b5732b/LICENSE","artifact_url":"https://raw.githubusercontent.com/BojarLab/glycowork/3d63f1ec25c850da3cde4d25cb602d50b6b5732b/LICENSE","version":"3d63f1ec25c850da3cde4d25cb602d50b6b5732b","retrieved_at":"2026-09-16T19:46:17.767851+00:00","artifact_sha256":"21d65f90ec86f746730748d9d53988b0936a1451642e8ec773c45380f3eb63d0","locator":"LICENSE","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-71e8ebc066164f0ce44e","kind":"source","name":"matchms/matchms: LICENSE","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/matchms/matchms/blob/066608589587c8d089afd2e8d55ceadb2766ea62/LICENSE","artifact_url":"https://raw.githubusercontent.com/matchms/matchms/066608589587c8d089afd2e8d55ceadb2766ea62/LICENSE","version":"066608589587c8d089afd2e8d55ceadb2766ea62","retrieved_at":"2026-09-16T19:46:19.674309+00:00","artifact_sha256":"38fcd5e9b1c63c25c5ca5ca2091dc66002be10ff2c78b0db71b80ef3b0d8335a","locator":"LICENSE","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-7267eabca7c4a7945878","kind":"source","name":"zhihan1996/DNABERT-2-117M: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://huggingface.co/zhihan1996/DNABERT-2-117M/blob/7bce263b15377fc15361f52cfab88f8b586abda0/README.md","artifact_url":"https://huggingface.co/zhihan1996/DNABERT-2-117M/blob/7bce263b15377fc15361f52cfab88f8b586abda0/README.md","version":"7bce263b15377fc15361f52cfab88f8b586abda0","retrieved_at":"2026-09-16T19:46:20.874956+00:00","artifact_sha256":"48b18abd051eb4952e0c0a50a0740387a1cea145be06517b67ab73a29431a3bc","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-72fdaed158e4ff851fbe","kind":"source","name":"biobakery/MetaPhlAn: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/biobakery/MetaPhlAn/blob/424f3e6e30618266404353e1083c6405a9f02f48/README.md","artifact_url":"https://raw.githubusercontent.com/biobakery/MetaPhlAn/424f3e6e30618266404353e1083c6405a9f02f48/README.md","version":"424f3e6e30618266404353e1083c6405a9f02f48","retrieved_at":"2026-09-16T19:46:18.561509+00:00","artifact_sha256":"ce491bb2d686145e0773c685d0d02e8a5fabc7daaea60eef07cf54298561fba7","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-734117ed8501eb26fefc","kind":"source","name":"jwohlwend/boltz: src/boltz/model/models/boltz2.py","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/jwohlwend/boltz/blob/b1ebfc46ecf57f5414e0d1a6f9027bbb122c53bc/src/boltz/model/models/boltz2.py","artifact_url":"https://raw.githubusercontent.com/jwohlwend/boltz/b1ebfc46ecf57f5414e0d1a6f9027bbb122c53bc/src/boltz/model/models/boltz2.py","version":"b1ebfc46ecf57f5414e0d1a6f9027bbb122c53bc","retrieved_at":"2026-09-16T19:46:19.465940+00:00","artifact_sha256":"f05169e66488910fc11c6a56b56d19a58e3f43649586218df75573eeff7a9945","locator":"src/boltz/model/models/boltz2.py","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-760ab2fa8c396aeb796c","kind":"source","name":"msalign: Primary paper PDF","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://arxiv.org/pdf/2605.19752","artifact_url":"https://arxiv.org/pdf/2605.19752","version":"2605.19752v1","retrieved_at":"2026-09-16T20:04:04.777728+00:00","artifact_sha256":"7395a55141ee7916741b7d9e6d4f42a1d3a03217a36ce6824ac3ff48afd4f26f","locator":"Primary paper PDF","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-790ab6283e5bb91474ed","kind":"source","name":"matsui-lab/GlycanGT: model/config_tokengt.py","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/matsui-lab/GlycanGT/blob/96611518c971deb89215ca163deaf9de3a59fa32/model/config_tokengt.py","artifact_url":"https://raw.githubusercontent.com/matsui-lab/GlycanGT/96611518c971deb89215ca163deaf9de3a59fa32/model/config_tokengt.py","version":"96611518c971deb89215ca163deaf9de3a59fa32","retrieved_at":"2026-09-16T19:46:19.737544+00:00","artifact_sha256":"355c055f44e4b09c33ed640606800014b411e96674a1cdb48157c6d57ea5d426","locator":"model/config_tokengt.py","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-7972ce2bdd6a5e40d8ad","kind":"source","name":"biomap-research/scFoundation: model/README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/biomap-research/scFoundation/blob/397631c495eddf9ad6644fc00c6ea8139e651245/model/README.md","artifact_url":"https://raw.githubusercontent.com/biomap-research/scFoundation/397631c495eddf9ad6644fc00c6ea8139e651245/model/README.md","version":"397631c495eddf9ad6644fc00c6ea8139e651245","retrieved_at":"2026-09-16T19:46:18.581063+00:00","artifact_sha256":"f62879233ecf5fac407cd516aca2ddb3df20370523cb111c9d057676b4721d5d","locator":"model/README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-7b6fdf915d9c6950ad0d","kind":"source","name":"dauparas/ProteinMPNN: training/README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/dauparas/ProteinMPNN/blob/8907e6671bfbfc92303b5f79c4b5e6ce47cdef57/training/README.md","artifact_url":"https://raw.githubusercontent.com/dauparas/ProteinMPNN/8907e6671bfbfc92303b5f79c4b5e6ce47cdef57/training/README.md","version":"8907e6671bfbfc92303b5f79c4b5e6ce47cdef57","retrieved_at":"2026-09-16T19:46:18.948060+00:00","artifact_sha256":"7cbe5f3f1cef53f1954b15710317d9755b7b7b3febace4758190850f12e23029","locator":"training/README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-7c43b700aa87c22843a6","kind":"source","name":"BojarLab/glycowork: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/BojarLab/glycowork/blob/3d63f1ec25c850da3cde4d25cb602d50b6b5732b/README.md","artifact_url":"https://raw.githubusercontent.com/BojarLab/glycowork/3d63f1ec25c850da3cde4d25cb602d50b6b5732b/README.md","version":"3d63f1ec25c850da3cde4d25cb602d50b6b5732b","retrieved_at":"2026-09-16T19:46:17.767851+00:00","artifact_sha256":"4d907a6723f3f56b14b13e35eeb0833ddf9a2f3512a1fafd07d79259105b264c","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-7cb8cb7f091536b92c11","kind":"source","name":"jwohlwend/boltz: scripts/train/configs/full.yaml","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/jwohlwend/boltz/blob/b1ebfc46ecf57f5414e0d1a6f9027bbb122c53bc/scripts/train/configs/full.yaml","artifact_url":"https://raw.githubusercontent.com/jwohlwend/boltz/b1ebfc46ecf57f5414e0d1a6f9027bbb122c53bc/scripts/train/configs/full.yaml","version":"b1ebfc46ecf57f5414e0d1a6f9027bbb122c53bc","retrieved_at":"2026-09-16T19:46:19.465940+00:00","artifact_sha256":"70aa005d0c29b1c28918e104b92fd43b5cfde6157993dfcd8fe832e9a2e107dc","locator":"scripts/train/configs/full.yaml","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-7e4b193e47ba209860a1","kind":"source","name":"instadeepai/nucleotide-transformer: LICENSE.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/instadeepai/nucleotide-transformer/blob/2dc37b86e16a6970fbc731751f7719d9f676f7f9/LICENSE.md","artifact_url":"https://raw.githubusercontent.com/instadeepai/nucleotide-transformer/2dc37b86e16a6970fbc731751f7719d9f676f7f9/LICENSE.md","version":"2dc37b86e16a6970fbc731751f7719d9f676f7f9","retrieved_at":"2026-09-16T19:46:19.364532+00:00","artifact_sha256":"1349a4b6148492b44f629e64eed676612e234fe9a839e4f3b277c1482c8849f1","locator":"LICENSE.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-7e59c59bfea79722fc33","kind":"source","name":"InstaDeepAI/segment_nt: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://huggingface.co/InstaDeepAI/segment_nt/blob/1048ad869036bb8e50a15d82b323ebeb52e8ab92/README.md","artifact_url":"https://huggingface.co/InstaDeepAI/segment_nt/blob/1048ad869036bb8e50a15d82b323ebeb52e8ab92/README.md","version":"1048ad869036bb8e50a15d82b323ebeb52e8ab92","retrieved_at":"2026-09-16T20:12:20.913226+00:00","artifact_sha256":"6e19e9c9293b0c392c053f92360b674b86c86c4457608b00c2fcb962bbd1d1bc","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-81e51077d7d3352a6de4","kind":"source","name":"soedinglab/MMseqs2: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/soedinglab/MMseqs2/blob/d401e78c2d18a822cdb1527d7464a043f6035a15/README.md","artifact_url":"https://raw.githubusercontent.com/soedinglab/MMseqs2/d401e78c2d18a822cdb1527d7464a043f6035a15/README.md","version":"d401e78c2d18a822cdb1527d7464a043f6035a15","retrieved_at":"2026-09-16T19:46:20.231620+00:00","artifact_sha256":"b6c591a763bf99c027857385f0e87ce5aa96caeaa74d71afd1fcec449eadb3d7","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-849b751ed4bf88622749","kind":"source","name":"ArcInstitute/evo2_7b: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://huggingface.co/ArcInstitute/evo2_7b/blob/bda0089f92582d5baabf0f22d9fc85f3588f6b58/README.md","artifact_url":"https://huggingface.co/ArcInstitute/evo2_7b/blob/bda0089f92582d5baabf0f22d9fc85f3588f6b58/README.md","version":"bda0089f92582d5baabf0f22d9fc85f3588f6b58","retrieved_at":"2026-09-16T20:04:02.231112+00:00","artifact_sha256":"802cb1e030bc560414a9fabcddd8fe243c296ad7d0b523dde0a3a0e1a7eaf794","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-863f6c2e6e18152c2f8b","kind":"source","name":"soedinglab/hh-suite: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/soedinglab/hh-suite/blob/43095e46ada4ec2a8a47d47ef5ad7e38b1429f7b/README.md","artifact_url":"https://raw.githubusercontent.com/soedinglab/hh-suite/43095e46ada4ec2a8a47d47ef5ad7e38b1429f7b/README.md","version":"43095e46ada4ec2a8a47d47ef5ad7e38b1429f7b","retrieved_at":"2026-09-16T19:46:20.402052+00:00","artifact_sha256":"2f8690d6a4a9767973ba9ea2af019d43d6108555ef0e9a69786033a71547d133","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-864a2c5ea6e61aa310f9","kind":"source","name":"Illumina/SpliceAI: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/Illumina/SpliceAI/blob/03f42437aaf56dc5dfd822c4ccee5aec1a705079/README.md","artifact_url":"https://raw.githubusercontent.com/Illumina/SpliceAI/03f42437aaf56dc5dfd822c4ccee5aec1a705079/README.md","version":"03f42437aaf56dc5dfd822c4ccee5aec1a705079","retrieved_at":"2026-09-16T19:46:17.769160+00:00","artifact_sha256":"8e5203afe343100832391e6155c7112f15cfe60bf0c21681d64e3420f854ef4d","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-867f2d1dcb4084248daa","kind":"source","name":"facebook/esmfold_v1: config.json","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://huggingface.co/facebook/esmfold_v1/blob/75a3841ee059df2bf4d56688166c8fb459ddd97a/config.json","artifact_url":"https://huggingface.co/facebook/esmfold_v1/blob/75a3841ee059df2bf4d56688166c8fb459ddd97a/config.json","version":"75a3841ee059df2bf4d56688166c8fb459ddd97a","retrieved_at":"2026-09-16T20:04:02.231029+00:00","artifact_sha256":"6b98125e2685fef2875499f6bd7c83968a077993ab53f99bf5581113665f7cc6","locator":"config.json","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-8717ffb993cc8f7eddec","kind":"source","name":"evo2: Journal full-text XML","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13128491/","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13128491/fullTextXML","version":"Retrieved page snapshot; no immutable publisher revision supplied","retrieved_at":"2026-09-16T20:16:14.422554+00:00","artifact_sha256":"d043dbda49e023ef6b70e36e7ca7832bda6b7af6c5934368e8778a7f23cda8bd","locator":"Journal full-text XML","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-87c70bddba97a72a5e5d","kind":"source","name":"glycangt: Journal full-text XML","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://europepmc.org/article/PMC/PMC13105845","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13105845/fullTextXML","version":"Primary article XML snapshot","retrieved_at":"2026-09-16T20:20:56.439056+00:00","artifact_sha256":"53e89a636c868c0329ee7eb6ae92f1028ec891940bc61730a148981b647fbbe5","locator":"Journal full-text XML","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-89de5de6fecdda12060d","kind":"source","name":"pluskal-lab/DreaMS: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/pluskal-lab/DreaMS/blob/dbec3a0b514a99e5056cfccde4559fda8cfe8129/README.md","artifact_url":"https://raw.githubusercontent.com/pluskal-lab/DreaMS/dbec3a0b514a99e5056cfccde4559fda8cfe8129/README.md","version":"dbec3a0b514a99e5056cfccde4559fda8cfe8129","retrieved_at":"2026-09-16T19:46:20.105311+00:00","artifact_sha256":"a6afe934f9894cc71fb9b561c7a122d043b377c27b3fc6f18505fe65b6226256","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-8a985eabfd054f0dec4d","kind":"source","name":"jwohlwend/boltz: LICENSE","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/jwohlwend/boltz/blob/b1ebfc46ecf57f5414e0d1a6f9027bbb122c53bc/LICENSE","artifact_url":"https://raw.githubusercontent.com/jwohlwend/boltz/b1ebfc46ecf57f5414e0d1a6f9027bbb122c53bc/LICENSE","version":"b1ebfc46ecf57f5414e0d1a6f9027bbb122c53bc","retrieved_at":"2026-09-16T19:46:19.465940+00:00","artifact_sha256":"f0667fd5e66c51e1ba8ddaa0249c6d7225b30037e02c45782d8f2c2943ac2617","locator":"LICENSE","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-8b4e1bd487c21cda9862","kind":"source","name":"cuhkaih/rhofold: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://huggingface.co/cuhkaih/rhofold/blob/4458c1c5484a3a10a7f3059b9f0e8ca0447b23ac/README.md","artifact_url":"https://huggingface.co/cuhkaih/rhofold/blob/4458c1c5484a3a10a7f3059b9f0e8ca0447b23ac/README.md","version":"4458c1c5484a3a10a7f3059b9f0e8ca0447b23ac","retrieved_at":"2026-09-16T20:04:02.231237+00:00","artifact_sha256":"f5672ffed6edaea59f16fe42dfa0bac09473801cd90cbcb6d8ca00995a2969c2","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-8e894fc6d180746a1808","kind":"source","name":"metagene-ai/METAGENE-1: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://huggingface.co/metagene-ai/METAGENE-1/blob/ad8a1e0ee62b85058bfc05d823d8e8d4759edc48/README.md","artifact_url":"https://huggingface.co/metagene-ai/METAGENE-1/blob/ad8a1e0ee62b85058bfc05d823d8e8d4759edc48/README.md","version":"ad8a1e0ee62b85058bfc05d823d8e8d4759edc48","retrieved_at":"2026-09-16T19:46:20.736359+00:00","artifact_sha256":"277517aa69527fccaeb85c124f3ec9772c5508fadb5c6ad423bee39a3071ad17","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-92b0ae57a020cbc40fe7","kind":"source","name":"proteinmpnn: Journal full-text XML","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC9997061/","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC9997061/fullTextXML","version":"Retrieved page snapshot; no immutable publisher revision supplied","retrieved_at":"2026-09-16T19:53:03.193318+00:00","artifact_sha256":"3e9042dc0ac2837e07654a43e74abcbcdd7dc4cf017fa81e2d9869b1fdb3c52e","locator":"Journal full-text XML","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-94c61d963d28fa3a8993","kind":"source","name":"https://meme-suite.org/meme/doc/copyright.html: copyright.html","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://meme-suite.org/meme/doc/copyright.html","artifact_url":"https://meme-suite.org/meme/doc/copyright.html","version":"Copyright 1994–2025; source snapshot 2026-09-16","retrieved_at":"2026-09-16T20:41:37.711422+00:00","artifact_sha256":"f54d004ea7c8e38222b76ffa4c04e072a204ccd948d41c3d46f55e36516b7262","locator":"copyright.html","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-950c72bc9d4bc037f6e2","kind":"source","name":"soedinglab/MMseqs2: LICENSE.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/soedinglab/MMseqs2/blob/d401e78c2d18a822cdb1527d7464a043f6035a15/LICENSE.md","artifact_url":"https://raw.githubusercontent.com/soedinglab/MMseqs2/d401e78c2d18a822cdb1527d7464a043f6035a15/LICENSE.md","version":"d401e78c2d18a822cdb1527d7464a043f6035a15","retrieved_at":"2026-09-16T19:46:20.231620+00:00","artifact_sha256":"adc3ea1f2f5096d2464460495e12a65a94f466edb9ccca50d1f25844ca83792a","locator":"LICENSE.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-955922130ff85a74fa18","kind":"source","name":"ml4bio/RhoFold: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/ml4bio/RhoFold/blob/6bdfbda720184409eb682ce08c05d258162ddc48/README.md","artifact_url":"https://raw.githubusercontent.com/ml4bio/RhoFold/6bdfbda720184409eb682ce08c05d258162ddc48/README.md","version":"6bdfbda720184409eb682ce08c05d258162ddc48","retrieved_at":"2026-09-16T19:46:19.891935+00:00","artifact_sha256":"530ee4a54cc39d30c7b66ddafe6f28a6162219812d977473a2d0ade7ad1fcba6","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-9576e3936b5a16a999cf","kind":"source","name":"matsui-lab/GlycanGT: LICENSE","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/matsui-lab/GlycanGT/blob/96611518c971deb89215ca163deaf9de3a59fa32/LICENSE","artifact_url":"https://raw.githubusercontent.com/matsui-lab/GlycanGT/96611518c971deb89215ca163deaf9de3a59fa32/LICENSE","version":"96611518c971deb89215ca163deaf9de3a59fa32","retrieved_at":"2026-09-16T19:46:19.737544+00:00","artifact_sha256":"c71d239df91726fc519c6eb72d318ec65820627232b2f796219e87dcf35d0ab4","locator":"LICENSE","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-95a620349435ae6651c5","kind":"source","name":"https://www.lipidmaps.org/resources/tools/lipidfinder/: page.html","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://www.lipidmaps.org/resources/tools/lipidfinder/","artifact_url":"https://www.lipidmaps.org/resources/tools/lipidfinder/","version":"Retrieved page snapshot; no immutable publisher revision supplied","retrieved_at":"2026-09-16T19:46:20.922105+00:00","artifact_sha256":"ae877d2c58b02e412782873bee4fc6216e8cb74e4497aa67f324e7f38a72e8ff","locator":"page.html","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-965600234c25fbd104c8","kind":"source","name":"neuralbioinfo/prokbert-mini: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://huggingface.co/neuralbioinfo/prokbert-mini/blob/feb2520a43cd9cdb5b3d8477e47209dbcb55d1dc/README.md","artifact_url":"https://huggingface.co/neuralbioinfo/prokbert-mini/blob/feb2520a43cd9cdb5b3d8477e47209dbcb55d1dc/README.md","version":"feb2520a43cd9cdb5b3d8477e47209dbcb55d1dc","retrieved_at":"2026-09-16T20:04:02.231162+00:00","artifact_sha256":"172ec5b600e342302df5a22d84b12e1ab61d05722adda549b6d82f65e5c5d658","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-96b5c3a31a50f7c259d6","kind":"source","name":"MAGICS-LAB/DNABERT_2: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/MAGICS-LAB/DNABERT_2/blob/f25bed9ee20db966dff39e5c1571249d04e36404/README.md","artifact_url":"https://raw.githubusercontent.com/MAGICS-LAB/DNABERT_2/f25bed9ee20db966dff39e5c1571249d04e36404/README.md","version":"f25bed9ee20db966dff39e5c1571249d04e36404","retrieved_at":"2026-09-16T19:46:17.892989+00:00","artifact_sha256":"734a8cec5f667d74d421bf3b273ad7e256216109636da45aa7ceba21cd34de16","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-97071500fc3422c426d4","kind":"source","name":"InstaDeepAI/segment_nt: config.json","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://huggingface.co/InstaDeepAI/segment_nt/blob/1048ad869036bb8e50a15d82b323ebeb52e8ab92/config.json","artifact_url":"https://huggingface.co/InstaDeepAI/segment_nt/blob/1048ad869036bb8e50a15d82b323ebeb52e8ab92/config.json","version":"1048ad869036bb8e50a15d82b323ebeb52e8ab92","retrieved_at":"2026-09-16T20:12:20.913226+00:00","artifact_sha256":"1412a48b99cfe79e6f34f84131ed964fd3ad5c4f3b3ea7db7ba2b0ac3466eadc","locator":"config.json","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-9863509ad10b824dc96b","kind":"source","name":"scfoundation-preprint: Primary paper PDF","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://yiheng-zhu.github.io/Yiheng/papers/5/scFoundation_bioRxiv_2023.pdf","artifact_url":"https://yiheng-zhu.github.io/Yiheng/papers/5/scFoundation_bioRxiv_2023.pdf","version":"bioRxiv manuscript posted 2023-06-15","retrieved_at":"2026-09-16T20:35:33.635910+00:00","artifact_sha256":"e2ce9d623a09b53eb863ae6ca359fe2956b7b0a7c3ebe7de6911aca3a660f737","locator":"Primary paper PDF","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-9a47f0326b33870b3113","kind":"source","name":"illumina/SpliceAI: LICENSE","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/illumina/SpliceAI/blob/03f42437aaf56dc5dfd822c4ccee5aec1a705079/LICENSE","artifact_url":"https://raw.githubusercontent.com/illumina/SpliceAI/03f42437aaf56dc5dfd822c4ccee5aec1a705079/LICENSE","version":"03f42437aaf56dc5dfd822c4ccee5aec1a705079","retrieved_at":"2026-09-16T19:46:19.364217+00:00","artifact_sha256":"67a909a0a8f8f7f45152207b6bcf9c78dd8a4dd3c8eef5bd11cd80a72e15344e","locator":"LICENSE","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-9b820532ba8e3965f64e","kind":"source","name":"illumina/SpliceAI: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/illumina/SpliceAI/blob/03f42437aaf56dc5dfd822c4ccee5aec1a705079/README.md","artifact_url":"https://raw.githubusercontent.com/illumina/SpliceAI/03f42437aaf56dc5dfd822c4ccee5aec1a705079/README.md","version":"03f42437aaf56dc5dfd822c4ccee5aec1a705079","retrieved_at":"2026-09-16T19:46:19.364217+00:00","artifact_sha256":"8e5203afe343100832391e6155c7112f15cfe60bf0c21681d64e3420f854ef4d","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-9ba2b6dc494c7b6df951","kind":"source","name":"DerrickWood/kraken2: docs/MANUAL.html","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/DerrickWood/kraken2/blob/8c190b1b668825935dbf6dee5f969227dc8269bb/docs/MANUAL.html","artifact_url":"https://raw.githubusercontent.com/DerrickWood/kraken2/8c190b1b668825935dbf6dee5f969227dc8269bb/docs/MANUAL.html","version":"8c190b1b668825935dbf6dee5f969227dc8269bb","retrieved_at":"2026-09-16T19:46:17.767980+00:00","artifact_sha256":"ef0962a736771bfb8373905d612d64f334544d625236f89e54cf3a04c9737016","locator":"docs/MANUAL.html","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-a07254268b1d1de1f7f8","kind":"source","name":"yeqinglin/genie3: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://huggingface.co/yeqinglin/genie3/blob/9ae31ebb8c56eebdc05ab282a8fd3f6a6d2a03a2/README.md","artifact_url":"https://huggingface.co/yeqinglin/genie3/blob/9ae31ebb8c56eebdc05ab282a8fd3f6a6d2a03a2/README.md","version":"9ae31ebb8c56eebdc05ab282a8fd3f6a6d2a03a2","retrieved_at":"2026-09-16T20:22:27.767036+00:00","artifact_sha256":"4bcf87ecfbbb8e07a01b21415a970c8b53a5283bf6872b657040d3f45c9241f7","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-a248990b44d481814f47","kind":"source","name":"ArcInstitute/evo2: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/ArcInstitute/evo2/blob/53f195997257c56c00e5ef8d33a54f5baad143a6/README.md","artifact_url":"https://raw.githubusercontent.com/ArcInstitute/evo2/53f195997257c56c00e5ef8d33a54f5baad143a6/README.md","version":"53f195997257c56c00e5ef8d33a54f5baad143a6","retrieved_at":"2026-09-16T19:46:17.765915+00:00","artifact_sha256":"58787c8ef5cb4fba4c04322a4ceb9f174e2233ec22d4193622fb6bc67d651d89","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-a48ae27aa000bdab7442","kind":"source","name":"zhihan1996/DNABERT-2-117M: LICENSE","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://huggingface.co/zhihan1996/DNABERT-2-117M/blob/7bce263b15377fc15361f52cfab88f8b586abda0/LICENSE","artifact_url":"https://huggingface.co/zhihan1996/DNABERT-2-117M/blob/7bce263b15377fc15361f52cfab88f8b586abda0/LICENSE","version":"7bce263b15377fc15361f52cfab88f8b586abda0","retrieved_at":"2026-09-16T19:46:20.874956+00:00","artifact_sha256":"c71d239df91726fc519c6eb72d318ec65820627232b2f796219e87dcf35d0ab4","locator":"LICENSE","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-a6b46962aae5cfc348c6","kind":"source","name":"metagene-ai/metagene-pretrain: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/metagene-ai/metagene-pretrain/blob/82b9e142db2c7e0a268346d53344d6c3bf223066/README.md","artifact_url":"https://raw.githubusercontent.com/metagene-ai/metagene-pretrain/82b9e142db2c7e0a268346d53344d6c3bf223066/README.md","version":"82b9e142db2c7e0a268346d53344d6c3bf223066","retrieved_at":"2026-09-16T20:04:02.447523+00:00","artifact_sha256":"f09b3bc7f21c19d31ad90205da47010dc73b75096c12af5b350a925b4cac5a7b","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-a766fb9176a491dc1ec3","kind":"source","name":"kundajelab/chrombpnet: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/kundajelab/chrombpnet/blob/09938fdb4397ec0006510e5251e48920a505d4de/README.md","artifact_url":"https://raw.githubusercontent.com/kundajelab/chrombpnet/09938fdb4397ec0006510e5251e48920a505d4de/README.md","version":"09938fdb4397ec0006510e5251e48920a505d4de","retrieved_at":"2026-09-16T19:46:19.514840+00:00","artifact_sha256":"058c4f98218ea2ed854681126b6f682c9f3beec91275781fb37e39c55d0c92ea","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-aadbeb0f10ec55d34f5f","kind":"source","name":"nt: Journal full-text XML","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11810778/","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11810778/fullTextXML","version":"Retrieved page snapshot; no immutable publisher revision supplied","retrieved_at":"2026-09-16T20:16:14.422628+00:00","artifact_sha256":"7c3b78a4f38ef053a08e466222e5662dd9711d91c535512d0b35d459d1fc7249","locator":"Journal full-text XML","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-ad7f5c1eb802d9895413","kind":"source","name":"aertslab/GENIE3: DESCRIPTION","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/aertslab/GENIE3/blob/54bc15636322e8773357de6e0b6683c6bc802825/DESCRIPTION","artifact_url":"https://raw.githubusercontent.com/aertslab/GENIE3/54bc15636322e8773357de6e0b6683c6bc802825/DESCRIPTION","version":"54bc15636322e8773357de6e0b6683c6bc802825","retrieved_at":"2026-09-16T19:46:18.166390+00:00","artifact_sha256":"4a2d48a4198b2ea6cbe5fa7749e9aa7286b318ebe19021bd1be52f29435e3196","locator":"DESCRIPTION","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-af4cadc1f0b4c69528d4","kind":"source","name":"aertslab/GRNBoost: LICENSE.txt","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/aertslab/GRNBoost/blob/26c836b3dcbb85852d3c6f4b8340e8655434da02/LICENSE.txt","artifact_url":"https://raw.githubusercontent.com/aertslab/GRNBoost/26c836b3dcbb85852d3c6f4b8340e8655434da02/LICENSE.txt","version":"26c836b3dcbb85852d3c6f4b8340e8655434da02","retrieved_at":"2026-09-16T19:46:18.179520+00:00","artifact_sha256":"16019e3b76d2c09bebcfb1186f59d35e40763142471ad79da82c87a8e27afaef","locator":"LICENSE.txt","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-af60e327735834e48fd7","kind":"source","name":"gcorso/DiffDock: LICENSE","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/gcorso/DiffDock/blob/85c49b60d3e0b0182a59ee43a34a6d7036981284/LICENSE","artifact_url":"https://raw.githubusercontent.com/gcorso/DiffDock/85c49b60d3e0b0182a59ee43a34a6d7036981284/LICENSE","version":"85c49b60d3e0b0182a59ee43a34a6d7036981284","retrieved_at":"2026-09-16T19:46:19.329065+00:00","artifact_sha256":"8efe3b8dac5c278ba290a66887e46261056f71dec75c2815c7da75451856cf1f","locator":"LICENSE","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-af8da643e933eae36a68","kind":"source","name":"polymathic-ai/MIMIC: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://huggingface.co/polymathic-ai/MIMIC/blob/72e63a1ece34928422fd46f89b6a6580ace99a97/README.md","artifact_url":"https://huggingface.co/polymathic-ai/MIMIC/blob/72e63a1ece34928422fd46f89b6a6580ace99a97/README.md","version":"72e63a1ece34928422fd46f89b6a6580ace99a97","retrieved_at":"2026-09-16T19:46:20.846683+00:00","artifact_sha256":"462acc643cba24d1d4c6f1419838402c5f3196e94260039baf61a59ef0a0a107","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-affbe2d511ff2a80f457","kind":"source","name":"chaidiscovery/chai-lab: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/chaidiscovery/chai-lab/blob/66c38d1fe5c6756a89ff8596b1dea87d305ec06f/README.md","artifact_url":"https://raw.githubusercontent.com/chaidiscovery/chai-lab/66c38d1fe5c6756a89ff8596b1dea87d305ec06f/README.md","version":"66c38d1fe5c6756a89ff8596b1dea87d305ec06f","retrieved_at":"2026-09-16T19:46:18.824295+00:00","artifact_sha256":"ea6f6e64f6fc73d0e3dcbe6755c2aab226fa13bad279d1213acc0217f3f5013f","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-b554679ea1503ce3d9f6","kind":"source","name":"metagene-ai/metagene-pretrain: train/config_hub/pretrain/genomicsllama.yml","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/metagene-ai/metagene-pretrain/blob/82b9e142db2c7e0a268346d53344d6c3bf223066/train/config_hub/pretrain/genomicsllama.yml","artifact_url":"https://raw.githubusercontent.com/metagene-ai/metagene-pretrain/82b9e142db2c7e0a268346d53344d6c3bf223066/train/config_hub/pretrain/genomicsllama.yml","version":"82b9e142db2c7e0a268346d53344d6c3bf223066","retrieved_at":"2026-09-16T20:04:02.447523+00:00","artifact_sha256":"0e764482ebf7e68c4751adfcf6430acdaa67fb683f59a31b9e25f8412d8ff5ad","locator":"train/config_hub/pretrain/genomicsllama.yml","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-b5e81b2c497214b18f8a","kind":"source","name":"snap-stanford/GEARS: gears/model.py","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/snap-stanford/GEARS/blob/f374e43e197b295016d80395d7a54ddb81cc6769/gears/model.py","artifact_url":"https://raw.githubusercontent.com/snap-stanford/GEARS/f374e43e197b295016d80395d7a54ddb81cc6769/gears/model.py","version":"f374e43e197b295016d80395d7a54ddb81cc6769","retrieved_at":"2026-09-16T19:46:20.191253+00:00","artifact_sha256":"07fee864bc5020807c90ed5443ff17530116f47f257332466243c233ef2bc857","locator":"gears/model.py","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-b8df803e8f55324a4eb5","kind":"source","name":"ArcInstitute/evo2_7b: config.json","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://huggingface.co/ArcInstitute/evo2_7b/blob/bda0089f92582d5baabf0f22d9fc85f3588f6b58/config.json","artifact_url":"https://huggingface.co/ArcInstitute/evo2_7b/blob/bda0089f92582d5baabf0f22d9fc85f3588f6b58/config.json","version":"bda0089f92582d5baabf0f22d9fc85f3588f6b58","retrieved_at":"2026-09-16T20:04:02.231112+00:00","artifact_sha256":"7f2e195e156de678b6d7db090dca19b37e88671f72ea7c9e1866e103946b69b2","locator":"config.json","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-b91fc1ec601eec6598c3","kind":"source","name":"chai1-web: Browser-extracted primary-paper passages","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://chaiassets.com/chai-1/paper/technical_report_v1.pdf","artifact_url":"https://chaiassets.com/chai-1/paper/technical_report_v1.pdf?trk=public_post_comment-text","version":"Technical report v1, 9 September 2024; browser-extracted passages pp.1–2,7–10","retrieved_at":"2026-09-16T20:25:06.247632+00:00","artifact_sha256":"8b6fcde51e45e97254308fead77ad8dae47d9990e5d6052221ce185644798608","locator":"Browser-extracted primary-paper passages","artifact_format":"browser_extracted_text","hash_scope":"SHA-256 of browser-extracted text artifact, not original PDF bytes; direct HTTP download returned403.","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-ba08b650243c9212de3f","kind":"source","name":"https://arxiv.org/abs/2605.19752: page.html","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://arxiv.org/abs/2605.19752","artifact_url":"https://arxiv.org/abs/2605.19752","version":"Retrieved page snapshot; no immutable publisher revision supplied","retrieved_at":"2026-09-16T19:46:17.765548+00:00","artifact_sha256":"e503f496841fd1d3836ebdab573de645257fc92fe7b5479056592da512e0ef3f","locator":"page.html","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-ba12a2d759aba6800ae6","kind":"source","name":"glycangt-supp: Publisher supplementary archive","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13105845/supplementaryFiles","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13105845/supplementaryFiles","version":"Retrieved page snapshot; no immutable publisher revision supplied","retrieved_at":"2026-09-16T20:42:10.495586+00:00","artifact_sha256":"17987fb07979d88c8f073905247bdacd22e4607e16c3c399bb292685e711846f","locator":"Publisher supplementary archive","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-bbfa62298e3bb7c5494d","kind":"source","name":"aqlaboratory/genie3: LICENSE","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/aqlaboratory/genie3/blob/d77ae5ac04212ff1e8b29b585859a3244c614804/LICENSE","artifact_url":"https://raw.githubusercontent.com/aqlaboratory/genie3/d77ae5ac04212ff1e8b29b585859a3244c614804/LICENSE","version":"d77ae5ac04212ff1e8b29b585859a3244c614804","retrieved_at":"2026-09-16T19:46:18.306022+00:00","artifact_sha256":"c71d239df91726fc519c6eb72d318ec65820627232b2f796219e87dcf35d0ab4","locator":"LICENSE","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-bc0327ef2ea6107e1773","kind":"source","name":"alphagenome-supp: Publisher supplementary archive","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12851941/supplementaryFiles","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12851941/supplementaryFiles","version":"Retrieved page snapshot; no immutable publisher revision supplied","retrieved_at":"2026-09-16T20:43:11.978411+00:00","artifact_sha256":"a006a3373cfdf42c41f9be7f64b883b38ea3f16be8037ea6024b5d0463f811dc","locator":"Publisher supplementary archive","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-bc3f1b0fa5b0999bbfa0","kind":"source","name":"kundajelab/chrombpnet: chrombpnet/training/models/chrombpnet_with_bias_model.py","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/kundajelab/chrombpnet/blob/09938fdb4397ec0006510e5251e48920a505d4de/chrombpnet/training/models/chrombpnet_with_bias_model.py","artifact_url":"https://raw.githubusercontent.com/kundajelab/chrombpnet/09938fdb4397ec0006510e5251e48920a505d4de/chrombpnet/training/models/chrombpnet_with_bias_model.py","version":"09938fdb4397ec0006510e5251e48920a505d4de","retrieved_at":"2026-09-16T19:46:19.514840+00:00","artifact_sha256":"bedc63a36ee27bb25ab39a00f1d9e3d4d64809754bf46aebe10d17b3a331dd33","locator":"chrombpnet/training/models/chrombpnet_with_bias_model.py","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-bd6ec1f819e22f075c32","kind":"source","name":"boltz2: Journal full-text XML","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://europepmc.org/article/PPR/PPR1039145","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PPR1039145/fullTextXML","version":"preprint; 1","retrieved_at":"2026-09-16T20:20:56.438787+00:00","artifact_sha256":"4ba7533096d594725e7f6362807aee6cc19596571fd09d9ae82fc49c6daa16fd","locator":"Journal full-text XML","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-bf1a5e03cd84873f4b04","kind":"source","name":"DerrickWood/kraken2: LICENSE","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/DerrickWood/kraken2/blob/8c190b1b668825935dbf6dee5f969227dc8269bb/LICENSE","artifact_url":"https://raw.githubusercontent.com/DerrickWood/kraken2/8c190b1b668825935dbf6dee5f969227dc8269bb/LICENSE","version":"8c190b1b668825935dbf6dee5f969227dc8269bb","retrieved_at":"2026-09-16T19:46:17.767980+00:00","artifact_sha256":"ef3803fed10bf0eae6919db5e12204af4460e73234953373b22bc6de04ed840a","locator":"LICENSE","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-bfbc8babf3630ddb3ed7","kind":"source","name":"biomap-research/scFoundation: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/biomap-research/scFoundation/blob/397631c495eddf9ad6644fc00c6ea8139e651245/README.md","artifact_url":"https://raw.githubusercontent.com/biomap-research/scFoundation/397631c495eddf9ad6644fc00c6ea8139e651245/README.md","version":"397631c495eddf9ad6644fc00c6ea8139e651245","retrieved_at":"2026-09-16T19:46:18.581063+00:00","artifact_sha256":"02a7ae44cf2cc9948b5c1520b18a261b45abf9f17e3149e37d9f48072184f376","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-c0351b621280a9c081da","kind":"source","name":"rfdiffusion: Journal full-text XML","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC10468394/","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC10468394/fullTextXML","version":"Retrieved page snapshot; no immutable publisher revision supplied","retrieved_at":"2026-09-16T20:16:14.422733+00:00","artifact_sha256":"bd41c7070976ae82e317a10367c81492a2692482c302a25c7f2a28ffe6ef4110","locator":"Journal full-text XML","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-c037e3419c04936262a0","kind":"source","name":"snap-stanford/GEARS: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/snap-stanford/GEARS/blob/f374e43e197b295016d80395d7a54ddb81cc6769/README.md","artifact_url":"https://raw.githubusercontent.com/snap-stanford/GEARS/f374e43e197b295016d80395d7a54ddb81cc6769/README.md","version":"f374e43e197b295016d80395d7a54ddb81cc6769","retrieved_at":"2026-09-16T19:46:20.191253+00:00","artifact_sha256":"fe78b7b1ede50c673b8f0d91fb2637707685c78b269ee157edb6cbc4d510d61b","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-c0ac4d10941b082e5f86","kind":"source","name":"aqlaboratory/openfold: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/aqlaboratory/openfold/blob/be2ec1841f16c966c65ae0e7599ebbadc725757d/README.md","artifact_url":"https://raw.githubusercontent.com/aqlaboratory/openfold/be2ec1841f16c966c65ae0e7599ebbadc725757d/README.md","version":"be2ec1841f16c966c65ae0e7599ebbadc725757d","retrieved_at":"2026-09-16T19:46:18.314650+00:00","artifact_sha256":"985b6dc09144ea14378dd2c543b288e8d2a05cb342a77cbc56b7a39c5170f638","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-c14c0e6a0fdbe1479658","kind":"source","name":"PolymathicAI/MIMIC: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/PolymathicAI/MIMIC/blob/9e652f16491e6c2e3881111e24b285c288554275/README.md","artifact_url":"https://raw.githubusercontent.com/PolymathicAI/MIMIC/9e652f16491e6c2e3881111e24b285c288554275/README.md","version":"9e652f16491e6c2e3881111e24b285c288554275","retrieved_at":"2026-09-16T20:04:02.231308+00:00","artifact_sha256":"694bed54356d27cf087890ed80649c08ef4520a0eb5c85374a7347138a6cbb2f","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-c2dd029568d234cb0d16","kind":"source","name":"metagene-ai/metagene-pretrain: train/LICENSE","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/metagene-ai/metagene-pretrain/blob/82b9e142db2c7e0a268346d53344d6c3bf223066/train/LICENSE","artifact_url":"https://raw.githubusercontent.com/metagene-ai/metagene-pretrain/82b9e142db2c7e0a268346d53344d6c3bf223066/train/LICENSE","version":"82b9e142db2c7e0a268346d53344d6c3bf223066","retrieved_at":"2026-09-16T20:04:02.447523+00:00","artifact_sha256":"ab89ceca19ca531005699095ca542f982fac92efd82f954f49ac17aa99a37cda","locator":"train/LICENSE","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-c303652f3d79132a00c2","kind":"source","name":"songlab-cal/tape: tape/models/modeling_bert.py","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/songlab-cal/tape/blob/6d345c2b2bbf52cd32cf179325c222afd92aec7e/tape/models/modeling_bert.py","artifact_url":"https://raw.githubusercontent.com/songlab-cal/tape/6d345c2b2bbf52cd32cf179325c222afd92aec7e/tape/models/modeling_bert.py","version":"6d345c2b2bbf52cd32cf179325c222afd92aec7e","retrieved_at":"2026-09-16T19:46:20.570146+00:00","artifact_sha256":"eba47206598a0e93e2f662d9c87343222556f236ac37373236e3970c43fb521d","locator":"tape/models/modeling_bert.py","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-c3d7a64294dcad7c6405","kind":"source","name":"google-deepmind/alphagenome_research: LICENSE","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/google-deepmind/alphagenome_research/blob/0db53bd4352c66d1e00a049a81da373a066e6670/LICENSE","artifact_url":"https://raw.githubusercontent.com/google-deepmind/alphagenome_research/0db53bd4352c66d1e00a049a81da373a066e6670/LICENSE","version":"0db53bd4352c66d1e00a049a81da373a066e6670","retrieved_at":"2026-09-16T19:46:19.361780+00:00","artifact_sha256":"cfc7749b96f63bd31c3c42b5c471bf756814053e847c10f3eb003417bc523d30","locator":"LICENSE","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-c465f6d4fcc04f9afe4b","kind":"source","name":"Illumina/SpliceAI: LICENSE","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/Illumina/SpliceAI/blob/03f42437aaf56dc5dfd822c4ccee5aec1a705079/LICENSE","artifact_url":"https://raw.githubusercontent.com/Illumina/SpliceAI/03f42437aaf56dc5dfd822c4ccee5aec1a705079/LICENSE","version":"03f42437aaf56dc5dfd822c4ccee5aec1a705079","retrieved_at":"2026-09-16T19:46:17.769160+00:00","artifact_sha256":"67a909a0a8f8f7f45152207b6bcf9c78dd8a4dd3c8eef5bd11cd80a72e15344e","locator":"LICENSE","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-c7befbb0fbd8620d5c3c","kind":"source","name":"RosettaCommons/RFdiffusion: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/RosettaCommons/RFdiffusion/blob/86507b6538f51fce57b5a72477165f03999ed7ae/README.md","artifact_url":"https://raw.githubusercontent.com/RosettaCommons/RFdiffusion/86507b6538f51fce57b5a72477165f03999ed7ae/README.md","version":"86507b6538f51fce57b5a72477165f03999ed7ae","retrieved_at":"2026-09-16T19:46:18.126662+00:00","artifact_sha256":"d89eb9790ce805ce23e1b1a6804f4d67c81056025e4852b875f8ddfd87a08125","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-c7d214a82cfd1afc827b","kind":"source","name":"aqlaboratory/openfold: LICENSE","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/aqlaboratory/openfold/blob/be2ec1841f16c966c65ae0e7599ebbadc725757d/LICENSE","artifact_url":"https://raw.githubusercontent.com/aqlaboratory/openfold/be2ec1841f16c966c65ae0e7599ebbadc725757d/LICENSE","version":"be2ec1841f16c966c65ae0e7599ebbadc725757d","retrieved_at":"2026-09-16T19:46:18.314650+00:00","artifact_sha256":"7a77a0f9b49715d3b124b39607c53c5df8795d26f7aae08b12e6dbb9b1ec1a40","locator":"LICENSE","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-d3d9afa5c6172982ed01","kind":"source","name":"ctheodoris/Geneformer: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://huggingface.co/ctheodoris/Geneformer/blob/1f7fbae4e469a5f4f1af8c111a529cfe1b3829f5/README.md","artifact_url":"https://huggingface.co/ctheodoris/Geneformer/blob/1f7fbae4e469a5f4f1af8c111a529cfe1b3829f5/README.md","version":"1f7fbae4e469a5f4f1af8c111a529cfe1b3829f5","retrieved_at":"2026-09-16T19:46:20.640731+00:00","artifact_sha256":"56d6e570b349cbedae9a54634421c94e7af8ea467ce0fdd79193372ae3cbdbd8","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-d65bf8ba587b6841e0d9","kind":"source","name":"biohub/ESMFold2: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://huggingface.co/biohub/ESMFold2/blob/69869f737beffec5294845ede23db5fc0b4f509e/README.md","artifact_url":"https://huggingface.co/biohub/ESMFold2/blob/69869f737beffec5294845ede23db5fc0b4f509e/README.md","version":"69869f737beffec5294845ede23db5fc0b4f509e","retrieved_at":"2026-09-16T20:22:27.767500+00:00","artifact_sha256":"7f20294b352b7e490689229c0fe26760115e3ffc66869d96dd89347729fe941d","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-d7e5c63feb0e0e627625","kind":"source","name":"facebook/esm2_t33_650M_UR50D: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://huggingface.co/facebook/esm2_t33_650M_UR50D/blob/08e4846e537177426273712802403f7ba8261b6c/README.md","artifact_url":"https://huggingface.co/facebook/esm2_t33_650M_UR50D/blob/08e4846e537177426273712802403f7ba8261b6c/README.md","version":"08e4846e537177426273712802403f7ba8261b6c","retrieved_at":"2026-09-16T20:04:02.230771+00:00","artifact_sha256":"462a2f24724e19c6be0efab926315c294a863c9a9770e2c8b3d859b2d81a07de","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-d8e2d4f2e752370b6565","kind":"source","name":"bowang-lab/scGPT: scgpt/model/model.py","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/bowang-lab/scGPT/blob/cebd6fae655b9c585a4807daa3ac31bb764f06b4/scgpt/model/model.py","artifact_url":"https://raw.githubusercontent.com/bowang-lab/scGPT/cebd6fae655b9c585a4807daa3ac31bb764f06b4/scgpt/model/model.py","version":"cebd6fae655b9c585a4807daa3ac31bb764f06b4","retrieved_at":"2026-09-16T19:46:18.594429+00:00","artifact_sha256":"4ae77618cc6f12a7d1f6d946fee3f7bd5c56f3e9e103073394a590ceba5a35ed","locator":"scgpt/model/model.py","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-da4566a88ed9fa42fdcb","kind":"source","name":"segmentnt: Journal full-text XML","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12615259/","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12615259/fullTextXML","version":"Retrieved page snapshot; no immutable publisher revision supplied","retrieved_at":"2026-09-16T20:16:14.422701+00:00","artifact_sha256":"b9bddff89c5f8898fc677008205a47b960ea569f3887cc15845e2547b92aaa70","locator":"Journal full-text XML","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-dacb012b22c0a7de4880","kind":"source","name":"Akikitani295/GlycanGT: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://huggingface.co/Akikitani295/GlycanGT/blob/7c9f4a19b366ca48ed03995a0ad47fa1e745f24c/README.md","artifact_url":"https://huggingface.co/Akikitani295/GlycanGT/blob/7c9f4a19b366ca48ed03995a0ad47fa1e745f24c/README.md","version":"7c9f4a19b366ca48ed03995a0ad47fa1e745f24c","retrieved_at":"2026-09-16T20:04:02.231272+00:00","artifact_sha256":"16a2967195f6be2d4478faf7151ceb7b96de4fa97ff168c7c975dc01d1dd69d3","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-dc53e8df56bb7352c014","kind":"source","name":"calico/basenji: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/calico/basenji/blob/06ce5d387e20b47184d05433b3983163c5f923cd/README.md","artifact_url":"https://raw.githubusercontent.com/calico/basenji/06ce5d387e20b47184d05433b3983163c5f923cd/README.md","version":"06ce5d387e20b47184d05433b3983163c5f923cd","retrieved_at":"2026-09-16T19:46:18.649671+00:00","artifact_sha256":"b77d93b4ff352874ecd21e074357b0bb7f4f09e7bfd53d8f7f152d05063c30c1","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-dd29c6cbb629171059a6","kind":"source","name":"tkzeng/Pangolin: LICENSE","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/tkzeng/Pangolin/blob/5cf94b8db938c658391b4305cd7ce33297d44ff7/LICENSE","artifact_url":"https://raw.githubusercontent.com/tkzeng/Pangolin/5cf94b8db938c658391b4305cd7ce33297d44ff7/LICENSE","version":"5cf94b8db938c658391b4305cd7ce33297d44ff7","retrieved_at":"2026-09-16T19:46:20.582143+00:00","artifact_sha256":"3972dc9744f6499f0f9b2dbf76696f2ae7ad8af9b23dde66d6af86c9dfb36986","locator":"LICENSE","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-de1ee43bdd6f04de9850","kind":"source","name":"pangolin: Journal full-text XML","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC9022248/","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC9022248/fullTextXML","version":"Retrieved page snapshot; no immutable publisher revision supplied","retrieved_at":"2026-09-16T19:53:03.008861+00:00","artifact_sha256":"c51d34f0bc17ffd34ee5c7caf4bbf627edd9f75f57497c8d6b133f405ec4c307","locator":"Journal full-text XML","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-ded281404bd1a0f3fdb7","kind":"source","name":"spliceai: Primary paper PDF","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://doi.org/10.1016/j.cell.2018.12.015","artifact_url":"https://www.marcottelab.org/users/BCH394P_364C_2021/SplicingAI-jaganathan2019.pdf","version":"Cell 2019 published article PDF, university-hosted copy","retrieved_at":"2026-09-16T20:04:38.697838+00:00","artifact_sha256":"e0d62bfd97c26907e184a6a262eb277e3ad0737322373df282c9832b56697dda","locator":"Primary paper PDF","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-e17e3864c464d981afc7","kind":"source","name":"ml4bio/RNA-FM: fm/pretrained.py","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/ml4bio/RNA-FM/blob/348951516e0963d22bbb33b3c9fc18c89081d38e/fm/pretrained.py","artifact_url":"https://raw.githubusercontent.com/ml4bio/RNA-FM/348951516e0963d22bbb33b3c9fc18c89081d38e/fm/pretrained.py","version":"348951516e0963d22bbb33b3c9fc18c89081d38e","retrieved_at":"2026-09-16T19:46:19.769679+00:00","artifact_sha256":"ab7271f8dbb876dbc5e2010e2d88b086088aada096f04149436d315a4f256aac","locator":"fm/pretrained.py","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-e508b40498c3a3ba53d5","kind":"source","name":"matchms/matchms: README.rst","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/matchms/matchms/blob/066608589587c8d089afd2e8d55ceadb2766ea62/README.rst","artifact_url":"https://raw.githubusercontent.com/matchms/matchms/066608589587c8d089afd2e8d55ceadb2766ea62/README.rst","version":"066608589587c8d089afd2e8d55ceadb2766ea62","retrieved_at":"2026-09-16T19:46:19.674309+00:00","artifact_sha256":"06a0c6b7adf444d7af53d20ce94c3cef483bd7be8db02fd8960c0704680d61ee","locator":"README.rst","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-e604b144dfbd97bf8917","kind":"source","name":"CAMI-challenge/CAMISIM: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/CAMI-challenge/CAMISIM/blob/7ce6013c6d5a0fac8ba8a80e52a03560d3546fa6/README.md","artifact_url":"https://raw.githubusercontent.com/CAMI-challenge/CAMISIM/7ce6013c6d5a0fac8ba8a80e52a03560d3546fa6/README.md","version":"7ce6013c6d5a0fac8ba8a80e52a03560d3546fa6","retrieved_at":"2026-09-16T19:46:17.767933+00:00","artifact_sha256":"ec5309bf16daab6c9a0adb393b191a3e716b9c627834cf3e08655c250a63da61","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-e69877a2a54345e162b3","kind":"source","name":"biohub/ESMC-6B: config.json","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://huggingface.co/biohub/ESMC-6B/blob/af1602ba7406f521b11bf8f81d52af378cde09e4/config.json","artifact_url":"https://huggingface.co/biohub/ESMC-6B/blob/af1602ba7406f521b11bf8f81d52af378cde09e4/config.json","version":"af1602ba7406f521b11bf8f81d52af378cde09e4","retrieved_at":"2026-09-16T20:22:27.767420+00:00","artifact_sha256":"80ccf981fbbd4f9f0a64830a0b64f1c79ed0f629d3f3b95f4424fd73174ac540","locator":"config.json","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-e847949b10d83e3b0efe","kind":"source","name":"songlab-cal/tape: LICENSE","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/songlab-cal/tape/blob/6d345c2b2bbf52cd32cf179325c222afd92aec7e/LICENSE","artifact_url":"https://raw.githubusercontent.com/songlab-cal/tape/6d345c2b2bbf52cd32cf179325c222afd92aec7e/LICENSE","version":"6d345c2b2bbf52cd32cf179325c222afd92aec7e","retrieved_at":"2026-09-16T19:46:20.570146+00:00","artifact_sha256":"b03177f56a7d17c782dfd8400ca4a29c272f46fd2ed48569d9194682f27e7001","locator":"LICENSE","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-ea568a52ea3ab476db6a","kind":"source","name":"dauparas/ProteinMPNN: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/dauparas/ProteinMPNN/blob/8907e6671bfbfc92303b5f79c4b5e6ce47cdef57/README.md","artifact_url":"https://raw.githubusercontent.com/dauparas/ProteinMPNN/8907e6671bfbfc92303b5f79c4b5e6ce47cdef57/README.md","version":"8907e6671bfbfc92303b5f79c4b5e6ce47cdef57","retrieved_at":"2026-09-16T19:46:18.948060+00:00","artifact_sha256":"772ebe52d2ba5100a28a888910c6f0c9fd4ded1d1372e3d89f6f1c48707e0365","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-eaf3efab8850672f44e4","kind":"source","name":"tkzeng/Pangolin: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/tkzeng/Pangolin/blob/5cf94b8db938c658391b4305cd7ce33297d44ff7/README.md","artifact_url":"https://raw.githubusercontent.com/tkzeng/Pangolin/5cf94b8db938c658391b4305cd7ce33297d44ff7/README.md","version":"5cf94b8db938c658391b4305cd7ce33297d44ff7","retrieved_at":"2026-09-16T19:46:20.582143+00:00","artifact_sha256":"9117f9d255d6b6e810d224a600d381192bccccd3f0cd417fed9559d49ca8fffd","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-eb328c1926fbe500f86b","kind":"source","name":"opencobra/cobrapy: README.rst","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/opencobra/cobrapy/blob/5aa19300fbf5dd632a9ac5c39ca28c1c621b943d/README.rst","artifact_url":"https://raw.githubusercontent.com/opencobra/cobrapy/5aa19300fbf5dd632a9ac5c39ca28c1c621b943d/README.rst","version":"5aa19300fbf5dd632a9ac5c39ca28c1c621b943d","retrieved_at":"2026-09-16T19:46:19.916278+00:00","artifact_sha256":"ae2da4ad3ae1d822e0010b6c023b26ef1d02b97007e17611dd4b1e22c70d4998","locator":"README.rst","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-ed0382dc026ea7080abd","kind":"source","name":"ArcInstitute/state: LICENSE","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/ArcInstitute/state/blob/9bbfe78a434a55205e4de834e1ea99f85f7a3add/LICENSE","artifact_url":"https://raw.githubusercontent.com/ArcInstitute/state/9bbfe78a434a55205e4de834e1ea99f85f7a3add/LICENSE","version":"9bbfe78a434a55205e4de834e1ea99f85f7a3add","retrieved_at":"2026-09-16T19:46:17.767683+00:00","artifact_sha256":"e66c269d4819aaab34b49ef5220c4ddab6756f21bb5180761a4eb8561f2b7bbd","locator":"LICENSE","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-edce0c06db05117706cb","kind":"source","name":"calico/basenji: docs/train.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/calico/basenji/blob/06ce5d387e20b47184d05433b3983163c5f923cd/docs/train.md","artifact_url":"https://raw.githubusercontent.com/calico/basenji/06ce5d387e20b47184d05433b3983163c5f923cd/docs/train.md","version":"06ce5d387e20b47184d05433b3983163c5f923cd","retrieved_at":"2026-09-16T19:46:18.649671+00:00","artifact_sha256":"2e42f0173b07e3493a570b901639a182b421c66c60741141278336b7757b76e6","locator":"docs/train.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-ee6e270af125ddf16eb3","kind":"source","name":"neuralbioinfo/prokbert-mini: config.json","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://huggingface.co/neuralbioinfo/prokbert-mini/blob/feb2520a43cd9cdb5b3d8477e47209dbcb55d1dc/config.json","artifact_url":"https://huggingface.co/neuralbioinfo/prokbert-mini/blob/feb2520a43cd9cdb5b3d8477e47209dbcb55d1dc/config.json","version":"feb2520a43cd9cdb5b3d8477e47209dbcb55d1dc","retrieved_at":"2026-09-16T20:04:02.231162+00:00","artifact_sha256":"6d68c922f919e0de3f87de60c07ebc1823927df806645937f94911c5b8245de6","locator":"config.json","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-eebe7c91156963e6ddc0","kind":"source","name":"dauparas/ProteinMPNN: LICENSE","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/dauparas/ProteinMPNN/blob/8907e6671bfbfc92303b5f79c4b5e6ce47cdef57/LICENSE","artifact_url":"https://raw.githubusercontent.com/dauparas/ProteinMPNN/8907e6671bfbfc92303b5f79c4b5e6ce47cdef57/LICENSE","version":"8907e6671bfbfc92303b5f79c4b5e6ce47cdef57","retrieved_at":"2026-09-16T19:46:18.948060+00:00","artifact_sha256":"82009d25ce585631f452b2b24589bdb29c559ccfefa2f200ef312ed5b501a586","locator":"LICENSE","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-efa6ec9d886a21534a20","kind":"source","name":"scgpt-may2023: Primary paper PDF","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://www.biorxiv.org/content/10.1101/2023.04.30.538439v1","artifact_url":"https://www.biorxiv.org/content/10.1101/2023.04.30.538439v1.full.pdf","version":"bioRxiv version posted 1 May 2023; re-inspected existing local research artifact","retrieved_at":"2026-09-16T20:21:23.046376+00:00","artifact_sha256":"4ef64d2b3431812df07c4b3e428950644ff264ec5728267c2e4692588f557419","locator":"Primary paper PDF","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Reused local archived primary PDF; no successful new publisher download claimed.","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-efc93d867cbf62e80ebb","kind":"source","name":"sokrypton/ColabFold: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/sokrypton/ColabFold/blob/84c27d9cc500489fd9b97545d2325b9d00f251d5/README.md","artifact_url":"https://raw.githubusercontent.com/sokrypton/ColabFold/84c27d9cc500489fd9b97545d2325b9d00f251d5/README.md","version":"84c27d9cc500489fd9b97545d2325b9d00f251d5","retrieved_at":"2026-09-16T19:46:20.488416+00:00","artifact_sha256":"a384ec62c86d568aeb0fe3e8a3b111f071fbcfacb937b407eb6da24fc70c823f","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-f06a8695b2915f86a45a","kind":"source","name":"snap-stanford/GEARS: LICENSE","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/snap-stanford/GEARS/blob/f374e43e197b295016d80395d7a54ddb81cc6769/LICENSE","artifact_url":"https://raw.githubusercontent.com/snap-stanford/GEARS/f374e43e197b295016d80395d7a54ddb81cc6769/LICENSE","version":"f374e43e197b295016d80395d7a54ddb81cc6769","retrieved_at":"2026-09-16T19:46:20.191253+00:00","artifact_sha256":"5c608f2c9421da729963328261ebffd23127dc55184fa8acc96acc52d8fd8b90","locator":"LICENSE","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-f078ec10fa45b5d77ad7","kind":"source","name":"aertslab/GENIE3: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/aertslab/GENIE3/blob/54bc15636322e8773357de6e0b6683c6bc802825/README.md","artifact_url":"https://raw.githubusercontent.com/aertslab/GENIE3/54bc15636322e8773357de6e0b6683c6bc802825/README.md","version":"54bc15636322e8773357de6e0b6683c6bc802825","retrieved_at":"2026-09-16T19:46:18.166390+00:00","artifact_sha256":"045aa63715acf615327d45ff42a897e77fa6d4b88084d2ad4e2ea836eb5fb48a","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-f0b17e30e69338b7762f","kind":"source","name":"https://meme-suite.org/meme/tools/fimo: page.html","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://meme-suite.org/meme/tools/fimo","artifact_url":"https://meme-suite.org/meme/tools/fimo","version":"Retrieved page snapshot; no immutable publisher revision supplied","retrieved_at":"2026-09-16T19:46:20.892034+00:00","artifact_sha256":"c0acb8f22d51b718d5c3a253e81b9882d804fa2aa153b6fcf3e826e22f085dd3","locator":"page.html","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-f29710879cf7a4438b12","kind":"source","name":"biobakery/humann: LICENSE","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/biobakery/humann/blob/e07b3a34d0b94c09a8ac5d28ff95009611178be2/LICENSE","artifact_url":"https://raw.githubusercontent.com/biobakery/humann/e07b3a34d0b94c09a8ac5d28ff95009611178be2/LICENSE","version":"e07b3a34d0b94c09a8ac5d28ff95009611178be2","retrieved_at":"2026-09-16T19:46:18.580457+00:00","artifact_sha256":"8b5d1a6cf9972029766b8909ad6d8a9459b2e6bab38d356b6a7953d850d74709","locator":"LICENSE","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-f3fd16387ba8459ecf34","kind":"source","name":"sokrypton/ColabFold: LICENSE","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/sokrypton/ColabFold/blob/84c27d9cc500489fd9b97545d2325b9d00f251d5/LICENSE","artifact_url":"https://raw.githubusercontent.com/sokrypton/ColabFold/84c27d9cc500489fd9b97545d2325b9d00f251d5/LICENSE","version":"84c27d9cc500489fd9b97545d2325b9d00f251d5","retrieved_at":"2026-09-16T19:46:20.488416+00:00","artifact_sha256":"f7a0429ad8a5ebc35994ba24d65a2962fc8c106ff28d2c7e43125ba4d2208108","locator":"LICENSE","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-f464cd407a6a9d031720","kind":"source","name":"gears: Journal full-text XML","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11180609/","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11180609/fullTextXML","version":"Retrieved page snapshot; no immutable publisher revision supplied","retrieved_at":"2026-09-16T20:16:14.422594+00:00","artifact_sha256":"6e9a8f4a2b8ccbc9aa46c43a34701b687f71b797048adf27aae2a99d9e2dcad9","locator":"Journal full-text XML","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-f60303c5eaa60ed9c77a","kind":"source","name":"ctheodoris/Geneformer: config.json","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://huggingface.co/ctheodoris/Geneformer/blob/1f7fbae4e469a5f4f1af8c111a529cfe1b3829f5/config.json","artifact_url":"https://huggingface.co/ctheodoris/Geneformer/blob/1f7fbae4e469a5f4f1af8c111a529cfe1b3829f5/config.json","version":"1f7fbae4e469a5f4f1af8c111a529cfe1b3829f5","retrieved_at":"2026-09-16T19:46:20.640731+00:00","artifact_sha256":"2cc4af3442644e84af71814a607b61c958d7b4e5ebde6791ab77b5f534ac6f6e","locator":"config.json","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-f6c94a04c51af984a008","kind":"source","name":"pluskal-lab/DreaMS: LICENSE","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/pluskal-lab/DreaMS/blob/dbec3a0b514a99e5056cfccde4559fda8cfe8129/LICENSE","artifact_url":"https://raw.githubusercontent.com/pluskal-lab/DreaMS/dbec3a0b514a99e5056cfccde4559fda8cfe8129/LICENSE","version":"dbec3a0b514a99e5056cfccde4559fda8cfe8129","retrieved_at":"2026-09-16T19:46:20.105311+00:00","artifact_sha256":"747bb9267b9d0e006a0486165ac2c9d5bd6481dfa7aefe808832703224af005d","locator":"LICENSE","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-f71e6c488025a4d82997","kind":"source","name":"metagene-ai/METAGENE-1: config.json","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://huggingface.co/metagene-ai/METAGENE-1/blob/ad8a1e0ee62b85058bfc05d823d8e8d4759edc48/config.json","artifact_url":"https://huggingface.co/metagene-ai/METAGENE-1/blob/ad8a1e0ee62b85058bfc05d823d8e8d4759edc48/config.json","version":"ad8a1e0ee62b85058bfc05d823d8e8d4759edc48","retrieved_at":"2026-09-16T19:46:20.736359+00:00","artifact_sha256":"dc3751f8648b4ab24a3c0c42026267e1a2561b93249f54b460c0374e62d33f98","locator":"config.json","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-f780521ccebbb0691f9e","kind":"source","name":"jwohlwend/boltz: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/jwohlwend/boltz/blob/b1ebfc46ecf57f5414e0d1a6f9027bbb122c53bc/README.md","artifact_url":"https://raw.githubusercontent.com/jwohlwend/boltz/b1ebfc46ecf57f5414e0d1a6f9027bbb122c53bc/README.md","version":"b1ebfc46ecf57f5414e0d1a6f9027bbb122c53bc","retrieved_at":"2026-09-16T19:46:19.465940+00:00","artifact_sha256":"79149435313841e6f9cc0c8b581259f3952a2469cbe6f9a733f7143b557025c2","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-fa53ab621979fb5f37c8","kind":"source","name":"facebook/esmfold_v1: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://huggingface.co/facebook/esmfold_v1/blob/75a3841ee059df2bf4d56688166c8fb459ddd97a/README.md","artifact_url":"https://huggingface.co/facebook/esmfold_v1/blob/75a3841ee059df2bf4d56688166c8fb459ddd97a/README.md","version":"75a3841ee059df2bf4d56688166c8fb459ddd97a","retrieved_at":"2026-09-16T20:04:02.231029+00:00","artifact_sha256":"12c535c3711211dcd3947ff8164d2e1428ca3461ef8a631ea7b6419673322fe2","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-fa7a6c37148223466dee","kind":"source","name":"gcorso/DiffDock: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/gcorso/DiffDock/blob/85c49b60d3e0b0182a59ee43a34a6d7036981284/README.md","artifact_url":"https://raw.githubusercontent.com/gcorso/DiffDock/85c49b60d3e0b0182a59ee43a34a6d7036981284/README.md","version":"85c49b60d3e0b0182a59ee43a34a6d7036981284","retrieved_at":"2026-09-16T19:46:19.329065+00:00","artifact_sha256":"6f63088d85b5f05d58416ede319387c1b7f3661b27741a36314ada861f2056de","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-fa91833592f376f6fb51","kind":"source","name":"RosettaCommons/RFdiffusion: LICENSE","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/RosettaCommons/RFdiffusion/blob/86507b6538f51fce57b5a72477165f03999ed7ae/LICENSE","artifact_url":"https://raw.githubusercontent.com/RosettaCommons/RFdiffusion/86507b6538f51fce57b5a72477165f03999ed7ae/LICENSE","version":"86507b6538f51fce57b5a72477165f03999ed7ae","retrieved_at":"2026-09-16T19:46:18.126662+00:00","artifact_sha256":"eefc2ae77cb92b1414a6ac76b246642fae2189747b1b07747e92e5ddc696ec24","locator":"LICENSE","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-fadad0bf451a225696d1","kind":"source","name":"https://fiehnlab.ucdavis.edu/projects/LipidBlast/: page.html","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://fiehnlab.ucdavis.edu/projects/LipidBlast/","artifact_url":"https://fiehnlab.ucdavis.edu/projects/LipidBlast/","version":"Retrieved page snapshot; no immutable publisher revision supplied","retrieved_at":"2026-09-16T19:46:17.765823+00:00","artifact_sha256":"584a7adeeb0f7ac33ca6f7434aaa16c945d2ababb53ea38b06e6cadde559880d","locator":"page.html","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-fd8e332abdf04a75195b","kind":"source","name":"ml4bio/RNA-FM: README.md","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/ml4bio/RNA-FM/blob/348951516e0963d22bbb33b3c9fc18c89081d38e/README.md","artifact_url":"https://raw.githubusercontent.com/ml4bio/RNA-FM/348951516e0963d22bbb33b3c9fc18c89081d38e/README.md","version":"348951516e0963d22bbb33b3c9fc18c89081d38e","retrieved_at":"2026-09-16T19:46:19.769679+00:00","artifact_sha256":"f9f1c1d62adc471661ca98b30c0250e9f3ce0cff7433830f149f5f48ea41c3da","locator":"README.md","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-fe7d7d8a007d344f589c","kind":"source","name":"calico/basenji: LICENSE","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/calico/basenji/blob/06ce5d387e20b47184d05433b3983163c5f923cd/LICENSE","artifact_url":"https://raw.githubusercontent.com/calico/basenji/06ce5d387e20b47184d05433b3983163c5f923cd/LICENSE","version":"06ce5d387e20b47184d05433b3983163c5f923cd","retrieved_at":"2026-09-16T19:46:18.649671+00:00","artifact_sha256":"3760a0dba24568a5b4a4e41e8aabe5e5e23d81422f26e7623c8ba5cf13ef6c05","locator":"LICENSE","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-official-fffd4026cb8f4018fe2e","kind":"source","name":"diffdockl: Primary paper PDF","description":"Primary-source artifact retrieved for field-level model-profile review. Source inspection is not independent model reproduction.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://arxiv.org/pdf/2402.18396","artifact_url":"https://arxiv.org/pdf/2402.18396","version":"2402.18396v1","retrieved_at":"2026-09-16T20:04:04.777388+00:00","artifact_sha256":"c8022ecbf5a2e628e42a03302a5595390020d40d65cc630490c5803ff6c14170","locator":"Primary paper PDF","artifact_format":"original_artifact","hash_scope":"SHA-256 of retrieved original artifact bytes","retrieval_note":"Direct source retrieval","review_method":"automated_source_review","review_scope":"Only cited profile claims; no numerical performance result added or independently reproduced."}} {"id":"evidence-reported-2ome-lm-2025-readme-md","kind":"source","name":"CSUBioGroup/2OMe-LM README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"fc90c421ffa9766c1981d6ea15eb73a9a2049f99858e1e3bef99f11263dee2bd","artifact_url":"https://raw.githubusercontent.com/CSUBioGroup/2OMe-LM/2e22439723777b5bacdce72cdcd7cfbde9e88cd1/README.md","retrieved_at":"2026-09-16T19:54:12.011441+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/CSUBioGroup/2OMe-LM/blob/2e22439723777b5bacdce72cdcd7cfbde9e88cd1/README.md","version":"2e22439723777b5bacdce72cdcd7cfbde9e88cd1"}} {"id":"evidence-reported-adar-gpt-editing-2026-license","kind":"source","name":"Scientific-Computing-Lab/ADAR-GPT LICENSE","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"02f64bae2cb1b1025702d2224fc7c0c762bdcfe58c9f83ec2d86f150f2f0fc3d","artifact_url":"https://raw.githubusercontent.com/Scientific-Computing-Lab/ADAR-GPT/c0fd23679922d91a45520455d4ca0202a5ca609f/LICENSE","retrieved_at":"2026-09-16T19:54:12.011620+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/Scientific-Computing-Lab/ADAR-GPT/blob/c0fd23679922d91a45520455d4ca0202a5ca609f/LICENSE","version":"c0fd23679922d91a45520455d4ca0202a5ca609f"}} {"id":"evidence-reported-adar-gpt-editing-2026-readme-md","kind":"source","name":"Scientific-Computing-Lab/ADAR-GPT README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"ae16bae98ce731633c412a0279285112ebe40db0f867907bd0caa91fce66ded8","artifact_url":"https://raw.githubusercontent.com/Scientific-Computing-Lab/ADAR-GPT/c0fd23679922d91a45520455d4ca0202a5ca609f/README.md","retrieved_at":"2026-09-16T19:54:12.011620+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/Scientific-Computing-Lab/ADAR-GPT/blob/c0fd23679922d91a45520455d4ca0202a5ca609f/README.md","version":"c0fd23679922d91a45520455d4ca0202a5ca609f"}} {"id":"evidence-reported-arsenal-regulatory-dna-2026-readme-md","kind":"source","name":"kundajelab/regulatory_lm README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"cae8dd0c9c9ba367844158a6474a961e7184633c6376c4fae81610054827218f","artifact_url":"https://raw.githubusercontent.com/kundajelab/regulatory_lm/2c4594f5b7649df3f1372238916f6d95bd31ba1d/README.md","retrieved_at":"2026-09-16T19:54:12.011707+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/kundajelab/regulatory_lm/blob/2c4594f5b7649df3f1372238916f6d95bd31ba1d/README.md","version":"2c4594f5b7649df3f1372238916f6d95bd31ba1d"}} {"id":"evidence-reported-barcodebert-2026-license","kind":"source","name":"bioscan-ml/BarcodeBERT LICENSE","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"6d98a262a5a27bc0671d802e51b9bd088a1020f64350a449e899533d153240a3","artifact_url":"https://raw.githubusercontent.com/bioscan-ml/BarcodeBERT/00e492374eb748ed0f034a3a5981ab4eeffd92cc/LICENSE","retrieved_at":"2026-09-16T19:54:12.012721+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/bioscan-ml/BarcodeBERT/blob/00e492374eb748ed0f034a3a5981ab4eeffd92cc/LICENSE","version":"00e492374eb748ed0f034a3a5981ab4eeffd92cc"}} {"id":"evidence-reported-barcodebert-2026-readme-md","kind":"source","name":"bioscan-ml/BarcodeBERT README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"540be344aa41f5b2740bed1819ccebdba6ce9681b7434c7fca02c6c1ce5ad360","artifact_url":"https://raw.githubusercontent.com/bioscan-ml/BarcodeBERT/00e492374eb748ed0f034a3a5981ab4eeffd92cc/README.md","retrieved_at":"2026-09-16T19:54:12.012721+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/bioscan-ml/BarcodeBERT/blob/00e492374eb748ed0f034a3a5981ab4eeffd92cc/README.md","version":"00e492374eb748ed0f034a3a5981ab4eeffd92cc"}} {"id":"evidence-reported-base-boltz-license","kind":"source","name":"jwohlwend/boltz LICENSE","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"f0667fd5e66c51e1ba8ddaa0249c6d7225b30037e02c45782d8f2c2943ac2617","artifact_url":"https://raw.githubusercontent.com/jwohlwend/boltz/b1ebfc46ecf57f5414e0d1a6f9027bbb122c53bc/LICENSE","retrieved_at":"2026-09-16T20:30:16.575839+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/jwohlwend/boltz/blob/b1ebfc46ecf57f5414e0d1a6f9027bbb122c53bc/LICENSE","version":"b1ebfc46ecf57f5414e0d1a6f9027bbb122c53bc"}} {"id":"evidence-reported-base-boltz-readme-md","kind":"source","name":"jwohlwend/boltz README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"79149435313841e6f9cc0c8b581259f3952a2469cbe6f9a733f7143b557025c2","artifact_url":"https://raw.githubusercontent.com/jwohlwend/boltz/b1ebfc46ecf57f5414e0d1a6f9027bbb122c53bc/README.md","retrieved_at":"2026-09-16T20:30:16.575839+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/jwohlwend/boltz/blob/b1ebfc46ecf57f5414e0d1a6f9027bbb122c53bc/README.md","version":"b1ebfc46ecf57f5414e0d1a6f9027bbb122c53bc"}} {"id":"evidence-reported-base-caduceus-license","kind":"source","name":"kuleshov-group/caduceus LICENSE","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"53c3bce42b068bd4c9a7831e18d4d7e7eab1b9cd00b8a3faac0aa96793c99bc5","artifact_url":"https://raw.githubusercontent.com/kuleshov-group/caduceus/0060a6d8079b6a040fc55d505e15972a327b70a6/LICENSE","retrieved_at":"2026-09-16T20:00:02.715892+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/kuleshov-group/caduceus/blob/0060a6d8079b6a040fc55d505e15972a327b70a6/LICENSE","version":"0060a6d8079b6a040fc55d505e15972a327b70a6"}} {"id":"evidence-reported-base-caduceus-readme-md","kind":"source","name":"kuleshov-group/caduceus README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"e508e5199d0cfb9c36dbd503cdccf50d734f419fa8cf1f890dec0db7e741ad42","artifact_url":"https://raw.githubusercontent.com/kuleshov-group/caduceus/0060a6d8079b6a040fc55d505e15972a327b70a6/README.md","retrieved_at":"2026-09-16T20:00:02.715892+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/kuleshov-group/caduceus/blob/0060a6d8079b6a040fc55d505e15972a327b70a6/README.md","version":"0060a6d8079b6a040fc55d505e15972a327b70a6"}} {"id":"evidence-reported-base-chai-license","kind":"source","name":"github.com/chaidiscovery/chai-lab LICENSE","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"511edf51c5c6f47bae9ae19c59d98666a682e0ccc98e90c2de2a9a897c44c003","artifact_url":"https://raw.githubusercontent.com/chaidiscovery/chai-lab/66c38d1fe5c6756a89ff8596b1dea87d305ec06f/LICENSE","retrieved_at":"2026-09-16T19:46:18.824295+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/chaidiscovery/chai-lab/blob/66c38d1fe5c6756a89ff8596b1dea87d305ec06f/LICENSE","version":"66c38d1fe5c6756a89ff8596b1dea87d305ec06f"}} {"id":"evidence-reported-base-chai-readme-md","kind":"source","name":"github.com/chaidiscovery/chai-lab README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"ea6f6e64f6fc73d0e3dcbe6755c2aab226fa13bad279d1213acc0217f3f5013f","artifact_url":"https://raw.githubusercontent.com/chaidiscovery/chai-lab/66c38d1fe5c6756a89ff8596b1dea87d305ec06f/README.md","retrieved_at":"2026-09-16T19:46:18.824295+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/chaidiscovery/chai-lab/blob/66c38d1fe5c6756a89ff8596b1dea87d305ec06f/README.md","version":"66c38d1fe5c6756a89ff8596b1dea87d305ec06f"}} {"id":"evidence-reported-base-diffdock-license","kind":"source","name":"gcorso/DiffDock LICENSE","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"8efe3b8dac5c278ba290a66887e46261056f71dec75c2815c7da75451856cf1f","artifact_url":"https://raw.githubusercontent.com/gcorso/DiffDock/85c49b60d3e0b0182a59ee43a34a6d7036981284/LICENSE","retrieved_at":"2026-09-16T20:00:00.818010+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/gcorso/DiffDock/blob/85c49b60d3e0b0182a59ee43a34a6d7036981284/LICENSE","version":"85c49b60d3e0b0182a59ee43a34a6d7036981284"}} {"id":"evidence-reported-base-diffdock-readme-md","kind":"source","name":"gcorso/DiffDock README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"6f63088d85b5f05d58416ede319387c1b7f3661b27741a36314ada861f2056de","artifact_url":"https://raw.githubusercontent.com/gcorso/DiffDock/85c49b60d3e0b0182a59ee43a34a6d7036981284/README.md","retrieved_at":"2026-09-16T20:00:00.818010+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/gcorso/DiffDock/blob/85c49b60d3e0b0182a59ee43a34a6d7036981284/README.md","version":"85c49b60d3e0b0182a59ee43a34a6d7036981284"}} {"id":"evidence-reported-base-dnabert2-card-license","kind":"source","name":"huggingface.co/zhihan1996/DNABERT-2-117M LICENSE","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"c71d239df91726fc519c6eb72d318ec65820627232b2f796219e87dcf35d0ab4","artifact_url":"https://huggingface.co/zhihan1996/DNABERT-2-117M/blob/7bce263b15377fc15361f52cfab88f8b586abda0/LICENSE","retrieved_at":"2026-09-16T19:46:20.874956+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://huggingface.co/zhihan1996/DNABERT-2-117M/blob/7bce263b15377fc15361f52cfab88f8b586abda0/LICENSE","version":"7bce263b15377fc15361f52cfab88f8b586abda0"}} {"id":"evidence-reported-base-dnabert2-license","kind":"source","name":"MAGICS-LAB/DNABERT_2 LICENSE","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"c71d239df91726fc519c6eb72d318ec65820627232b2f796219e87dcf35d0ab4","artifact_url":"https://raw.githubusercontent.com/MAGICS-LAB/DNABERT_2/f25bed9ee20db966dff39e5c1571249d04e36404/LICENSE","retrieved_at":"2026-09-16T20:00:02.624261+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/MAGICS-LAB/DNABERT_2/blob/f25bed9ee20db966dff39e5c1571249d04e36404/LICENSE","version":"f25bed9ee20db966dff39e5c1571249d04e36404"}} {"id":"evidence-reported-base-dnabert2-readme-md","kind":"source","name":"MAGICS-LAB/DNABERT_2 README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"734a8cec5f667d74d421bf3b273ad7e256216109636da45aa7ceba21cd34de16","artifact_url":"https://raw.githubusercontent.com/MAGICS-LAB/DNABERT_2/f25bed9ee20db966dff39e5c1571249d04e36404/README.md","retrieved_at":"2026-09-16T20:00:02.624261+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/MAGICS-LAB/DNABERT_2/blob/f25bed9ee20db966dff39e5c1571249d04e36404/README.md","version":"f25bed9ee20db966dff39e5c1571249d04e36404"}} {"id":"evidence-reported-base-esm-license","kind":"source","name":"facebookresearch/esm LICENSE","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"da6d3703ed11cbe42bd212c725957c98da23cbff1998c05fa4b3d976d1a58e93","artifact_url":"https://raw.githubusercontent.com/facebookresearch/esm/2b369911bb5b4b0dda914521b9475cad1656b2ac/LICENSE","retrieved_at":"2026-09-16T20:00:00.816433+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/facebookresearch/esm/blob/2b369911bb5b4b0dda914521b9475cad1656b2ac/LICENSE","version":"2b369911bb5b4b0dda914521b9475cad1656b2ac"}} {"id":"evidence-reported-base-esm-readme-md","kind":"source","name":"facebookresearch/esm README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"8b273c21a322fc9473d1b68d0dd40c8166ab2f89e4a190aa26ca87251b97cba9","artifact_url":"https://raw.githubusercontent.com/facebookresearch/esm/2b369911bb5b4b0dda914521b9475cad1656b2ac/README.md","retrieved_at":"2026-09-16T20:00:00.816433+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/facebookresearch/esm/blob/2b369911bb5b4b0dda914521b9475cad1656b2ac/README.md","version":"2b369911bb5b4b0dda914521b9475cad1656b2ac"}} {"id":"evidence-reported-base-evo2-license","kind":"source","name":"ArcInstitute/evo2 LICENSE","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"5bb5812fc2bfb2d777fe5621767172f8ef30be62ad719c87e0f10832336a99e0","artifact_url":"https://raw.githubusercontent.com/ArcInstitute/evo2/53f195997257c56c00e5ef8d33a54f5baad143a6/LICENSE","retrieved_at":"2026-09-16T20:00:02.431871+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/ArcInstitute/evo2/blob/53f195997257c56c00e5ef8d33a54f5baad143a6/LICENSE","version":"53f195997257c56c00e5ef8d33a54f5baad143a6"}} {"id":"evidence-reported-base-evo2-readme-md","kind":"source","name":"ArcInstitute/evo2 README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"58787c8ef5cb4fba4c04322a4ceb9f174e2233ec22d4193622fb6bc67d651d89","artifact_url":"https://raw.githubusercontent.com/ArcInstitute/evo2/53f195997257c56c00e5ef8d33a54f5baad143a6/README.md","retrieved_at":"2026-09-16T20:00:02.431871+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/ArcInstitute/evo2/blob/53f195997257c56c00e5ef8d33a54f5baad143a6/README.md","version":"53f195997257c56c00e5ef8d33a54f5baad143a6"}} {"id":"evidence-reported-base-geneformer-readme-md","kind":"source","name":"huggingface.co/ctheodoris/Geneformer README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"56d6e570b349cbedae9a54634421c94e7af8ea467ce0fdd79193372ae3cbdbd8","artifact_url":"https://huggingface.co/ctheodoris/Geneformer/blob/1f7fbae4e469a5f4f1af8c111a529cfe1b3829f5/README.md","retrieved_at":"2026-09-16T19:46:20.640731+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://huggingface.co/ctheodoris/Geneformer/blob/1f7fbae4e469a5f4f1af8c111a529cfe1b3829f5/README.md","version":"1f7fbae4e469a5f4f1af8c111a529cfe1b3829f5"}} {"id":"evidence-reported-base-genomad-license","kind":"source","name":"apcamargo/genomad LICENSE","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"2878b1cb109d8dde17ea2607f6def6e3225b003caccf1dba98ad7233e331db87","artifact_url":"https://raw.githubusercontent.com/apcamargo/genomad/8c5fd0d1722d458a3e8ff50278cdc00ed4a514fc/LICENSE","retrieved_at":"2026-09-16T20:00:02.618496+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/apcamargo/genomad/blob/8c5fd0d1722d458a3e8ff50278cdc00ed4a514fc/LICENSE","version":"8c5fd0d1722d458a3e8ff50278cdc00ed4a514fc"}} {"id":"evidence-reported-base-genomad-readme-md","kind":"source","name":"apcamargo/genomad README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"0d7ab49689a8b402b2e081cecf38e6fe2fb44bccbbd2b8bd61bb08aaa53720af","artifact_url":"https://raw.githubusercontent.com/apcamargo/genomad/8c5fd0d1722d458a3e8ff50278cdc00ed4a514fc/README.md","retrieved_at":"2026-09-16T20:00:02.618496+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/apcamargo/genomad/blob/8c5fd0d1722d458a3e8ff50278cdc00ed4a514fc/README.md","version":"8c5fd0d1722d458a3e8ff50278cdc00ed4a514fc"}} {"id":"evidence-reported-base-gtdbtk-license","kind":"source","name":"Ecogenomics/GTDBTk LICENSE","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"12ac5047f2af0522f06798b1589ffc4599bc29c91f954d7874e0320634e777c0","artifact_url":"https://raw.githubusercontent.com/Ecogenomics/GTDBTk/f17decef1f9d9cf5b4d31fd21f5c9d32d813abdc/LICENSE","retrieved_at":"2026-09-16T20:00:03.977720+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/Ecogenomics/GTDBTk/blob/f17decef1f9d9cf5b4d31fd21f5c9d32d813abdc/LICENSE","version":"f17decef1f9d9cf5b4d31fd21f5c9d32d813abdc"}} {"id":"evidence-reported-base-gtdbtk-readme-md","kind":"source","name":"Ecogenomics/GTDBTk README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"72124a5fb40dc02379ca775daae2bf894bc9e1b240a0a31d65e6c6c5c99e268c","artifact_url":"https://raw.githubusercontent.com/Ecogenomics/GTDBTk/f17decef1f9d9cf5b4d31fd21f5c9d32d813abdc/README.md","retrieved_at":"2026-09-16T20:00:03.977720+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/Ecogenomics/GTDBTk/blob/f17decef1f9d9cf5b4d31fd21f5c9d32d813abdc/README.md","version":"f17decef1f9d9cf5b4d31fd21f5c9d32d813abdc"}} {"id":"evidence-reported-base-hyenadna-license","kind":"source","name":"HazyResearch/hyena-dna LICENSE","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"c71d239df91726fc519c6eb72d318ec65820627232b2f796219e87dcf35d0ab4","artifact_url":"https://raw.githubusercontent.com/HazyResearch/hyena-dna/d553021b483b82980aa4b868b37ec2d4332e198a/LICENSE","retrieved_at":"2026-09-16T20:00:02.742030+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/HazyResearch/hyena-dna/blob/d553021b483b82980aa4b868b37ec2d4332e198a/LICENSE","version":"d553021b483b82980aa4b868b37ec2d4332e198a"}} {"id":"evidence-reported-base-hyenadna-readme-md","kind":"source","name":"HazyResearch/hyena-dna README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"2b551789c76b8a552c1d989b194ef72a4cd83790d06bc2a8ac2af6ebee92e8db","artifact_url":"https://raw.githubusercontent.com/HazyResearch/hyena-dna/d553021b483b82980aa4b868b37ec2d4332e198a/README.md","retrieved_at":"2026-09-16T20:00:02.742030+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/HazyResearch/hyena-dna/blob/d553021b483b82980aa4b868b37ec2d4332e198a/README.md","version":"d553021b483b82980aa4b868b37ec2d4332e198a"}} {"id":"evidence-reported-base-kraken2-license","kind":"source","name":"DerrickWood/kraken2 LICENSE","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"ef3803fed10bf0eae6919db5e12204af4460e73234953373b22bc6de04ed840a","artifact_url":"https://raw.githubusercontent.com/DerrickWood/kraken2/8c190b1b668825935dbf6dee5f969227dc8269bb/LICENSE","retrieved_at":"2026-09-16T20:00:00.821204+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/DerrickWood/kraken2/blob/8c190b1b668825935dbf6dee5f969227dc8269bb/LICENSE","version":"8c190b1b668825935dbf6dee5f969227dc8269bb"}} {"id":"evidence-reported-base-kraken2-readme-md","kind":"source","name":"DerrickWood/kraken2 README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"2ea33af266b4268a55fd750d0f3265cd3165d61f6be375c5ea3b3ff5c58c7c8c","artifact_url":"https://raw.githubusercontent.com/DerrickWood/kraken2/8c190b1b668825935dbf6dee5f969227dc8269bb/README.md","retrieved_at":"2026-09-16T20:00:00.821204+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/DerrickWood/kraken2/blob/8c190b1b668825935dbf6dee5f969227dc8269bb/README.md","version":"8c190b1b668825935dbf6dee5f969227dc8269bb"}} {"id":"evidence-reported-base-metaphlan-license-txt","kind":"source","name":"biobakery/MetaPhlAn license.txt","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"ecf18c2928e49997ba1f098f0f0da0958d257d69fffb48875162dbd24f3d7762","artifact_url":"https://raw.githubusercontent.com/biobakery/MetaPhlAn/424f3e6e30618266404353e1083c6405a9f02f48/license.txt","retrieved_at":"2026-09-16T20:00:02.332125+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/biobakery/MetaPhlAn/blob/424f3e6e30618266404353e1083c6405a9f02f48/license.txt","version":"424f3e6e30618266404353e1083c6405a9f02f48"}} {"id":"evidence-reported-base-metaphlan-readme-md","kind":"source","name":"biobakery/MetaPhlAn README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"ce491bb2d686145e0773c685d0d02e8a5fabc7daaea60eef07cf54298561fba7","artifact_url":"https://raw.githubusercontent.com/biobakery/MetaPhlAn/424f3e6e30618266404353e1083c6405a9f02f48/README.md","retrieved_at":"2026-09-16T20:00:02.332125+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/biobakery/MetaPhlAn/blob/424f3e6e30618266404353e1083c6405a9f02f48/README.md","version":"424f3e6e30618266404353e1083c6405a9f02f48"}} {"id":"evidence-reported-base-pangolin-license","kind":"source","name":"github.com/tkzeng/Pangolin LICENSE","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"3972dc9744f6499f0f9b2dbf76696f2ae7ad8af9b23dde66d6af86c9dfb36986","artifact_url":"https://raw.githubusercontent.com/tkzeng/Pangolin/5cf94b8db938c658391b4305cd7ce33297d44ff7/LICENSE","retrieved_at":"2026-09-16T19:46:20.582143+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/tkzeng/Pangolin/blob/5cf94b8db938c658391b4305cd7ce33297d44ff7/LICENSE","version":"5cf94b8db938c658391b4305cd7ce33297d44ff7"}} {"id":"evidence-reported-base-pangolin-readme-md","kind":"source","name":"github.com/tkzeng/Pangolin README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"9117f9d255d6b6e810d224a600d381192bccccd3f0cd417fed9559d49ca8fffd","artifact_url":"https://raw.githubusercontent.com/tkzeng/Pangolin/5cf94b8db938c658391b4305cd7ce33297d44ff7/README.md","retrieved_at":"2026-09-16T19:46:20.582143+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/tkzeng/Pangolin/blob/5cf94b8db938c658391b4305cd7ce33297d44ff7/README.md","version":"5cf94b8db938c658391b4305cd7ce33297d44ff7"}} {"id":"evidence-reported-base-promotech-license","kind":"source","name":"BioinformaticsLabAtMUN/Promotech LICENSE","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"3972dc9744f6499f0f9b2dbf76696f2ae7ad8af9b23dde66d6af86c9dfb36986","artifact_url":"https://raw.githubusercontent.com/BioinformaticsLabAtMUN/Promotech/56251ad9b883ef831b4753fc623d5ec970fe65e0/LICENSE","retrieved_at":"2026-09-16T20:40:28.842295+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/BioinformaticsLabAtMUN/Promotech/blob/56251ad9b883ef831b4753fc623d5ec970fe65e0/LICENSE","version":"56251ad9b883ef831b4753fc623d5ec970fe65e0"}} {"id":"evidence-reported-base-promotech-readme-md","kind":"source","name":"BioinformaticsLabAtMUN/Promotech README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"51050c0bf5982be985dd87e450ea5c2408a2d87cd32c679fc639246d64e5ca80","artifact_url":"https://raw.githubusercontent.com/BioinformaticsLabAtMUN/Promotech/56251ad9b883ef831b4753fc623d5ec970fe65e0/README.md","retrieved_at":"2026-09-16T20:40:28.842295+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/BioinformaticsLabAtMUN/Promotech/blob/56251ad9b883ef831b4753fc623d5ec970fe65e0/README.md","version":"56251ad9b883ef831b4753fc623d5ec970fe65e0"}} {"id":"evidence-reported-base-proteinmpnn-license","kind":"source","name":"dauparas/ProteinMPNN LICENSE","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"82009d25ce585631f452b2b24589bdb29c559ccfefa2f200ef312ed5b501a586","artifact_url":"https://raw.githubusercontent.com/dauparas/ProteinMPNN/8907e6671bfbfc92303b5f79c4b5e6ce47cdef57/LICENSE","retrieved_at":"2026-09-16T20:00:02.940818+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/dauparas/ProteinMPNN/blob/8907e6671bfbfc92303b5f79c4b5e6ce47cdef57/LICENSE","version":"8907e6671bfbfc92303b5f79c4b5e6ce47cdef57"}} {"id":"evidence-reported-base-proteinmpnn-readme-md","kind":"source","name":"dauparas/ProteinMPNN README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"772ebe52d2ba5100a28a888910c6f0c9fd4ded1d1372e3d89f6f1c48707e0365","artifact_url":"https://raw.githubusercontent.com/dauparas/ProteinMPNN/8907e6671bfbfc92303b5f79c4b5e6ce47cdef57/README.md","retrieved_at":"2026-09-16T20:00:02.940818+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/dauparas/ProteinMPNN/blob/8907e6671bfbfc92303b5f79c4b5e6ce47cdef57/README.md","version":"8907e6671bfbfc92303b5f79c4b5e6ce47cdef57"}} {"id":"evidence-reported-base-rinalmo-license","kind":"source","name":"lbcb-sci/RiNALMo LICENSE","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"c71d239df91726fc519c6eb72d318ec65820627232b2f796219e87dcf35d0ab4","artifact_url":"https://raw.githubusercontent.com/lbcb-sci/RiNALMo/2c2c5c14a5ae609d8c560a5d9ca32e51e0288955/LICENSE","retrieved_at":"2026-09-16T20:00:00.817876+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/lbcb-sci/RiNALMo/blob/2c2c5c14a5ae609d8c560a5d9ca32e51e0288955/LICENSE","version":"2c2c5c14a5ae609d8c560a5d9ca32e51e0288955"}} {"id":"evidence-reported-base-rinalmo-readme-md","kind":"source","name":"lbcb-sci/RiNALMo README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"de3d60dd5aedc63c4f401a92876c2d1e053011e008773ce1be38511360463b63","artifact_url":"https://raw.githubusercontent.com/lbcb-sci/RiNALMo/2c2c5c14a5ae609d8c560a5d9ca32e51e0288955/README.md","retrieved_at":"2026-09-16T20:00:00.817876+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/lbcb-sci/RiNALMo/blob/2c2c5c14a5ae609d8c560a5d9ca32e51e0288955/README.md","version":"2c2c5c14a5ae609d8c560a5d9ca32e51e0288955"}} {"id":"evidence-reported-base-rnafm-license","kind":"source","name":"ml4bio/RNA-FM LICENSE","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"b0809e99b532fdf51660f5a3d2a9010ed09d15aef0d131ad80fe80c2291a4fba","artifact_url":"https://raw.githubusercontent.com/ml4bio/RNA-FM/348951516e0963d22bbb33b3c9fc18c89081d38e/LICENSE","retrieved_at":"2026-09-16T20:00:00.817676+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/ml4bio/RNA-FM/blob/348951516e0963d22bbb33b3c9fc18c89081d38e/LICENSE","version":"348951516e0963d22bbb33b3c9fc18c89081d38e"}} {"id":"evidence-reported-base-rnafm-readme-md","kind":"source","name":"ml4bio/RNA-FM README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"f9f1c1d62adc471661ca98b30c0250e9f3ce0cff7433830f149f5f48ea41c3da","artifact_url":"https://raw.githubusercontent.com/ml4bio/RNA-FM/348951516e0963d22bbb33b3c9fc18c89081d38e/README.md","retrieved_at":"2026-09-16T20:00:00.817676+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/ml4bio/RNA-FM/blob/348951516e0963d22bbb33b3c9fc18c89081d38e/README.md","version":"348951516e0963d22bbb33b3c9fc18c89081d38e"}} {"id":"evidence-reported-base-rnafold-license-txt","kind":"source","name":"ViennaRNA/ViennaRNA license.txt","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"04ca785cab24944ad8ea6a2ddfea47b91246806c937ee65b5cc33f32c9dd897d","artifact_url":"https://raw.githubusercontent.com/ViennaRNA/ViennaRNA/1ffec79f5e258896160f7362ced8263450f371dc/license.txt","retrieved_at":"2026-09-16T20:00:00.820987+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/ViennaRNA/ViennaRNA/blob/1ffec79f5e258896160f7362ced8263450f371dc/license.txt","version":"1ffec79f5e258896160f7362ced8263450f371dc"}} {"id":"evidence-reported-base-rnafold-readme-md","kind":"source","name":"ViennaRNA/ViennaRNA README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"d37146b01e4273062a5230c496a9af8414ee7ef14fcd905cf415959256d59c4e","artifact_url":"https://raw.githubusercontent.com/ViennaRNA/ViennaRNA/1ffec79f5e258896160f7362ced8263450f371dc/README.md","retrieved_at":"2026-09-16T20:00:00.820987+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/ViennaRNA/ViennaRNA/blob/1ffec79f5e258896160f7362ced8263450f371dc/README.md","version":"1ffec79f5e258896160f7362ced8263450f371dc"}} {"id":"evidence-reported-base-scgpt-license","kind":"source","name":"bowang-lab/scGPT LICENSE","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"1ceeacbed51e2890187425547bc2efd16c1b7ad45189b7dfb21e83a45a2e9d9e","artifact_url":"https://raw.githubusercontent.com/bowang-lab/scGPT/cebd6fae655b9c585a4807daa3ac31bb764f06b4/LICENSE","retrieved_at":"2026-09-16T20:00:00.816587+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/bowang-lab/scGPT/blob/cebd6fae655b9c585a4807daa3ac31bb764f06b4/LICENSE","version":"cebd6fae655b9c585a4807daa3ac31bb764f06b4"}} {"id":"evidence-reported-base-scgpt-readme-md","kind":"source","name":"bowang-lab/scGPT README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"b0503e8ca789f19f1fc2350c5aaf57b1b323bbae43b354655231b5f4a1586c83","artifact_url":"https://raw.githubusercontent.com/bowang-lab/scGPT/cebd6fae655b9c585a4807daa3ac31bb764f06b4/README.md","retrieved_at":"2026-09-16T20:00:00.816587+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/bowang-lab/scGPT/blob/cebd6fae655b9c585a4807daa3ac31bb764f06b4/README.md","version":"cebd6fae655b9c585a4807daa3ac31bb764f06b4"}} {"id":"evidence-reported-base-scvi-license","kind":"source","name":"scverse/scvi-tools LICENSE","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"66399db0284d2539790efb348886ab0c1f745bbe5fea6ac38a00465a14adc8f5","artifact_url":"https://raw.githubusercontent.com/scverse/scvi-tools/73b28e44223621470e582a81a102c107bb22678b/LICENSE","retrieved_at":"2026-09-16T20:00:02.370441+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/scverse/scvi-tools/blob/73b28e44223621470e582a81a102c107bb22678b/LICENSE","version":"73b28e44223621470e582a81a102c107bb22678b"}} {"id":"evidence-reported-base-scvi-readme-md","kind":"source","name":"scverse/scvi-tools README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"eb46b8a54e60643ca0cd8cb375ede05b01dcbd2380ca17ce8027a92ba13cebbb","artifact_url":"https://raw.githubusercontent.com/scverse/scvi-tools/73b28e44223621470e582a81a102c107bb22678b/README.md","retrieved_at":"2026-09-16T20:00:02.370441+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/scverse/scvi-tools/blob/73b28e44223621470e582a81a102c107bb22678b/README.md","version":"73b28e44223621470e582a81a102c107bb22678b"}} {"id":"evidence-reported-base-spliceai-license","kind":"source","name":"github.com/Illumina/SpliceAI LICENSE","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"67a909a0a8f8f7f45152207b6bcf9c78dd8a4dd3c8eef5bd11cd80a72e15344e","artifact_url":"https://raw.githubusercontent.com/Illumina/SpliceAI/03f42437aaf56dc5dfd822c4ccee5aec1a705079/LICENSE","retrieved_at":"2026-09-16T19:46:17.769160+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/Illumina/SpliceAI/blob/03f42437aaf56dc5dfd822c4ccee5aec1a705079/LICENSE","version":"03f42437aaf56dc5dfd822c4ccee5aec1a705079"}} {"id":"evidence-reported-base-spliceai-readme-md","kind":"source","name":"github.com/Illumina/SpliceAI README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"8e5203afe343100832391e6155c7112f15cfe60bf0c21681d64e3420f854ef4d","artifact_url":"https://raw.githubusercontent.com/Illumina/SpliceAI/03f42437aaf56dc5dfd822c4ccee5aec1a705079/README.md","retrieved_at":"2026-09-16T19:46:17.769160+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/Illumina/SpliceAI/blob/03f42437aaf56dc5dfd822c4ccee5aec1a705079/README.md","version":"03f42437aaf56dc5dfd822c4ccee5aec1a705079"}} {"id":"evidence-reported-base-ufold-license","kind":"source","name":"uci-cbcl/UFold LICENSE","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"5c93b3a2d31b6e7b964633ff5e81bd9e31ff9ee845996ab3b0d3d71b0d368f2e","artifact_url":"https://raw.githubusercontent.com/uci-cbcl/UFold/75bd9acc83826059682dfca9d3659df66b132cd1/LICENSE","retrieved_at":"2026-09-16T20:00:03.913031+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/uci-cbcl/UFold/blob/75bd9acc83826059682dfca9d3659df66b132cd1/LICENSE","version":"75bd9acc83826059682dfca9d3659df66b132cd1"}} {"id":"evidence-reported-base-ufold-readme-md","kind":"source","name":"uci-cbcl/UFold README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"9ef66e737e1047f7028f8f636516f5d0144f9395fe5aceb1a387d1ba457c81f2","artifact_url":"https://raw.githubusercontent.com/uci-cbcl/UFold/75bd9acc83826059682dfca9d3659df66b132cd1/README.md","retrieved_at":"2026-09-16T20:00:03.913031+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/uci-cbcl/UFold/blob/75bd9acc83826059682dfca9d3659df66b132cd1/README.md","version":"75bd9acc83826059682dfca9d3659df66b132cd1"}} {"id":"evidence-reported-base-vina-license","kind":"source","name":"ccsb-scripps/AutoDock-Vina LICENSE","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"cfc7749b96f63bd31c3c42b5c471bf756814053e847c10f3eb003417bc523d30","artifact_url":"https://raw.githubusercontent.com/ccsb-scripps/AutoDock-Vina/3c65c0b3e6c2c1d183f6a175ecb65e3c5ba91645/LICENSE","retrieved_at":"2026-09-16T20:00:01.500953+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/ccsb-scripps/AutoDock-Vina/blob/3c65c0b3e6c2c1d183f6a175ecb65e3c5ba91645/LICENSE","version":"3c65c0b3e6c2c1d183f6a175ecb65e3c5ba91645"}} {"id":"evidence-reported-base-vina-readme-md","kind":"source","name":"ccsb-scripps/AutoDock-Vina README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"4f1728521ab79de1c33e1cf8605b31037effed5de2a2fbbccba58d7b0a005ae7","artifact_url":"https://raw.githubusercontent.com/ccsb-scripps/AutoDock-Vina/3c65c0b3e6c2c1d183f6a175ecb65e3c5ba91645/README.md","retrieved_at":"2026-09-16T20:00:01.500953+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/ccsb-scripps/AutoDock-Vina/blob/3c65c0b3e6c2c1d183f6a175ecb65e3c5ba91645/README.md","version":"3c65c0b3e6c2c1d183f6a175ecb65e3c5ba91645"}} {"id":"evidence-reported-birna-bert-2025-readme-md","kind":"source","name":"buetnlpbio/BiRNA-BERT README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"00f3835cae38a2307d8095fc5240496e01a9a981c8450230f9ae5722153ac2cc","artifact_url":"https://raw.githubusercontent.com/buetnlpbio/BiRNA-BERT/14dc86b1b44c266f01025fc425103f2878646b39/README.md","retrieved_at":"2026-09-16T19:54:12.014900+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/buetnlpbio/BiRNA-BERT/blob/14dc86b1b44c266f01025fc425103f2878646b39/README.md","version":"14dc86b1b44c266f01025fc425103f2878646b39"}} {"id":"evidence-reported-bpfold-2025-license","kind":"source","name":"heqin-zhu/BPfold LICENSE","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"dd82fc4d63d005354cb0f1eab0d3f80248586dcfdaac4286c700787ca28cf141","artifact_url":"https://raw.githubusercontent.com/heqin-zhu/BPfold/d37d6aa10cbca13e590ff83917fc4d63fec2ddbc/LICENSE","retrieved_at":"2026-09-16T19:54:12.015212+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/heqin-zhu/BPfold/blob/d37d6aa10cbca13e590ff83917fc4d63fec2ddbc/LICENSE","version":"d37d6aa10cbca13e590ff83917fc4d63fec2ddbc"}} {"id":"evidence-reported-bpfold-2025-readme-md","kind":"source","name":"heqin-zhu/BPfold README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"e3d23006a2a2e4524208bdc4561d3091aeaad1f08c1e79ce2b23549b0ffbdd51","artifact_url":"https://raw.githubusercontent.com/heqin-zhu/BPfold/d37d6aa10cbca13e590ff83917fc4d63fec2ddbc/README.md","retrieved_at":"2026-09-16T19:54:12.015212+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/heqin-zhu/BPfold/blob/d37d6aa10cbca13e590ff83917fc4d63fec2ddbc/README.md","version":"d37d6aa10cbca13e590ff83917fc4d63fec2ddbc"}} {"id":"evidence-reported-bpfold-supplement","kind":"source","name":"BPfold supplementary information","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_member":"41467_2025_60048_MOESM1_ESM.pdf","artifact_sha256":"254d4842e4690d21a565833aa0e27c7d45cc1ef926dd52527596cc171be91e64","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12216785/supplementaryFiles","retrieved_at":"2026-09-16T20:39:46.798664+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12216785/supplementaryFiles","version":"s41467-025-60048-1 published supplement"}} {"id":"evidence-reported-cammiq-2022-license","kind":"source","name":"algo-cancer/CAMMiQ LICENSE","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"92dd25a63adf6299964d45946d39a1d6e47a12957b8c60570b94df58510aa535","artifact_url":"https://raw.githubusercontent.com/algo-cancer/CAMMiQ/6142150d427a74cc21a5ee4d8b37a3b78884f163/LICENSE","retrieved_at":"2026-09-16T19:54:12.016679+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/algo-cancer/CAMMiQ/blob/6142150d427a74cc21a5ee4d8b37a3b78884f163/LICENSE","version":"6142150d427a74cc21a5ee4d8b37a3b78884f163"}} {"id":"evidence-reported-cammiq-2022-readme-md","kind":"source","name":"algo-cancer/CAMMiQ README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"94b01aa6c1d8657fe6d6c77367aeb70991557d6dcd33e8229acc931bf5ce6b3c","artifact_url":"https://raw.githubusercontent.com/algo-cancer/CAMMiQ/6142150d427a74cc21a5ee4d8b37a3b78884f163/README.md","retrieved_at":"2026-09-16T19:54:12.016679+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/algo-cancer/CAMMiQ/blob/6142150d427a74cc21a5ee4d8b37a3b78884f163/README.md","version":"6142150d427a74cc21a5ee4d8b37a3b78884f163"}} {"id":"evidence-reported-cathe2-2025-license","kind":"source","name":"Mouret-Orfeu/CATHe2 LICENSE","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"71ecefd54a74530222dbfa19f602ca441c4042183cec725a4c4db44b556eaef0","artifact_url":"https://raw.githubusercontent.com/Mouret-Orfeu/CATHe2/cdaf5f4c8d9dfda86f78fd0efc83ab22a30e9ffd/LICENSE","retrieved_at":"2026-09-16T20:30:16.571174+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/Mouret-Orfeu/CATHe2/blob/cdaf5f4c8d9dfda86f78fd0efc83ab22a30e9ffd/LICENSE","version":"cdaf5f4c8d9dfda86f78fd0efc83ab22a30e9ffd"}} {"id":"evidence-reported-cathe2-2025-readme-md","kind":"source","name":"Mouret-Orfeu/CATHe2 README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"8e7bd5bd27851c9b96bb7edadeee6dac9ab5cc1913e831f2234055ccb0b5f779","artifact_url":"https://raw.githubusercontent.com/Mouret-Orfeu/CATHe2/cdaf5f4c8d9dfda86f78fd0efc83ab22a30e9ffd/README.md","retrieved_at":"2026-09-16T20:30:16.571174+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/Mouret-Orfeu/CATHe2/blob/cdaf5f4c8d9dfda86f78fd0efc83ab22a30e9ffd/README.md","version":"cdaf5f4c8d9dfda86f78fd0efc83ab22a30e9ffd"}} {"id":"evidence-reported-cell-dino-2025-docs-readme-cell-dino-md","kind":"source","name":"facebookresearch/dinov2 docs/README_CELL_DINO.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"b9ec5771aaa41d571f5c2b54300a7f8bf7ab5d75f7679d2a80f260247a71d42f","artifact_url":"https://raw.githubusercontent.com/facebookresearch/dinov2/7764ea0f912e53c92e82eb78a2a1631e92725fc8/docs/README_CELL_DINO.md","retrieved_at":"2026-09-16T20:30:16.572226+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/facebookresearch/dinov2/blob/7764ea0f912e53c92e82eb78a2a1631e92725fc8/docs/README_CELL_DINO.md","version":"7764ea0f912e53c92e82eb78a2a1631e92725fc8"}} {"id":"evidence-reported-cell-dino-2025-license-cell-dino-code","kind":"source","name":"facebookresearch/dinov2 LICENSE_CELL_DINO_CODE","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"fe7b4ce83b8381cc5b216bbb4af73c570688d1b819c73bbaed8ca401f4677cd6","artifact_url":"https://raw.githubusercontent.com/facebookresearch/dinov2/7764ea0f912e53c92e82eb78a2a1631e92725fc8/LICENSE_CELL_DINO_CODE","retrieved_at":"2026-09-16T20:30:16.572226+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/facebookresearch/dinov2/blob/7764ea0f912e53c92e82eb78a2a1631e92725fc8/LICENSE_CELL_DINO_CODE","version":"7764ea0f912e53c92e82eb78a2a1631e92725fc8"}} {"id":"evidence-reported-cell-dino-2025-license-cell-dino-models","kind":"source","name":"facebookresearch/dinov2 LICENSE_CELL_DINO_MODELS","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"d779b47a4ef8bbfc6c90d14768fe6a0a2c1c08bbc3712084496058bfd83cef4f","artifact_url":"https://raw.githubusercontent.com/facebookresearch/dinov2/7764ea0f912e53c92e82eb78a2a1631e92725fc8/LICENSE_CELL_DINO_MODELS","retrieved_at":"2026-09-16T20:30:16.572226+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/facebookresearch/dinov2/blob/7764ea0f912e53c92e82eb78a2a1631e92725fc8/LICENSE_CELL_DINO_MODELS","version":"7764ea0f912e53c92e82eb78a2a1631e92725fc8"}} {"id":"evidence-reported-cell2sentence-2024-license","kind":"source","name":"vandijklab/cell2sentence LICENSE","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"cbb25571ab70c6ebeebb0068243283a887a9353808543d535fedcd882be30fd6","artifact_url":"https://raw.githubusercontent.com/vandijklab/cell2sentence/a6efaf079f98491d4723ced44b929936b94368aa/LICENSE","retrieved_at":"2026-09-16T20:42:56.481821+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/vandijklab/cell2sentence/blob/a6efaf079f98491d4723ced44b929936b94368aa/LICENSE","version":"a6efaf079f98491d4723ced44b929936b94368aa"}} {"id":"evidence-reported-cell2sentence-2024-readme-md","kind":"source","name":"vandijklab/cell2sentence README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"60b183e5311a46eff139d728d860875874ff255242bcd7a237789b893ffc262c","artifact_url":"https://raw.githubusercontent.com/vandijklab/cell2sentence/a6efaf079f98491d4723ced44b929936b94368aa/README.md","retrieved_at":"2026-09-16T20:42:56.481821+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/vandijklab/cell2sentence/blob/a6efaf079f98491d4723ced44b929936b94368aa/README.md","version":"a6efaf079f98491d4723ced44b929936b94368aa"}} {"id":"evidence-reported-cobra-rna-binding-2026-license","kind":"source","name":"kucm-lsbi/CoBRA LICENSE","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"58dbf5d5eba9fda0cb32f823379e9a5ed549a31308f36bf72fa33c747e0fbcb9","artifact_url":"https://raw.githubusercontent.com/kucm-lsbi/CoBRA/415fd05cabf990f28a46cc2ba651531a28f7d249/LICENSE","retrieved_at":"2026-09-16T19:54:13.479891+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/kucm-lsbi/CoBRA/blob/415fd05cabf990f28a46cc2ba651531a28f7d249/LICENSE","version":"415fd05cabf990f28a46cc2ba651531a28f7d249"}} {"id":"evidence-reported-cobra-rna-binding-2026-readme-md","kind":"source","name":"kucm-lsbi/CoBRA README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"9ad9fea6b42049ca4ca3f71dd906e6d1289ef842aee14bb4241e880866003bc9","artifact_url":"https://raw.githubusercontent.com/kucm-lsbi/CoBRA/415fd05cabf990f28a46cc2ba651531a28f7d249/README.md","retrieved_at":"2026-09-16T19:54:13.479891+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/kucm-lsbi/CoBRA/blob/415fd05cabf990f28a46cc2ba651531a28f7d249/README.md","version":"415fd05cabf990f28a46cc2ba651531a28f7d249"}} {"id":"evidence-reported-codonbert-vaccines-2024-readme-md","kind":"source","name":"Sanofi-Public/CodonBert README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"9efbe98650ca68f2d4797776df178dce59489fdac6beb6fa7d8390f74dfe9eb1","artifact_url":"https://raw.githubusercontent.com/Sanofi-Public/CodonBert/451a1b167c06028dfbf2ff7aa2cfdea46fbcc4f4/README.md","retrieved_at":"2026-09-16T19:54:13.645687+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/Sanofi-Public/CodonBert/blob/451a1b167c06028dfbf2ff7aa2cfdea46fbcc4f4/README.md","version":"451a1b167c06028dfbf2ff7aa2cfdea46fbcc4f4"}} {"id":"evidence-reported-cupid-rna-interactions-2026-license-txt","kind":"source","name":"AnacletoLAB/ncRNA-CUPID LICENSE.txt","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"3376f2209385e0f99839898bebb8cb39edb37ce7c23121c0b5501b99dab6ca40","artifact_url":"https://raw.githubusercontent.com/AnacletoLAB/ncRNA-CUPID/f663c10d2f6c33f8513614badbbc673e654a64d8/LICENSE.txt","retrieved_at":"2026-09-16T19:54:13.686121+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/AnacletoLAB/ncRNA-CUPID/blob/f663c10d2f6c33f8513614badbbc673e654a64d8/LICENSE.txt","version":"f663c10d2f6c33f8513614badbbc673e654a64d8"}} {"id":"evidence-reported-cupid-rna-interactions-2026-readme-md","kind":"source","name":"AnacletoLAB/ncRNA-CUPID README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"17e3dbece27b0cd7b6031ecc817f941ad351c95ee0416e0bd8e4020ecaed7f5b","artifact_url":"https://raw.githubusercontent.com/AnacletoLAB/ncRNA-CUPID/f663c10d2f6c33f8513614badbbc673e654a64d8/README.md","retrieved_at":"2026-09-16T19:54:13.686121+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/AnacletoLAB/ncRNA-CUPID/blob/f663c10d2f6c33f8513614badbbc673e654a64d8/README.md","version":"f663c10d2f6c33f8513614badbbc673e654a64d8"}} {"id":"evidence-reported-cyaprombert-2022-readme-md","kind":"source","name":"hanepira/TSSnote-CyaPromBert README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"c916c6c99b07fa3a440264eeefeac8a6b1508867da31c7233bc4d24367fd8bbd","artifact_url":"https://raw.githubusercontent.com/hanepira/TSSnote-CyaPromBert/e86f5449e2e2af3fead1b418ba721f38feb61318/README.md","retrieved_at":"2026-09-16T19:54:13.729708+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/hanepira/TSSnote-CyaPromBert/blob/e86f5449e2e2af3fead1b418ba721f38feb61318/README.md","version":"e86f5449e2e2af3fead1b418ba721f38feb61318"}} {"id":"evidence-reported-deelig-2021-readme-md","kind":"source","name":"asadahmedtech/DEELIG README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"a6c75971df142bd46ea22287f97aeafd861b035548cddef2bd8ad75f75f3aafd","artifact_url":"https://raw.githubusercontent.com/asadahmedtech/DEELIG/3a3993fc903c40f1ce904111c8e085c79fb45df6/README.md","retrieved_at":"2026-09-16T19:54:13.759450+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/asadahmedtech/DEELIG/blob/3a3993fc903c40f1ce904111c8e085c79fb45df6/README.md","version":"3a3993fc903c40f1ce904111c8e085c79fb45df6"}} {"id":"evidence-reported-deepinteraware-2025-readme-md","kind":"source","name":"BioMedicalBigDataMiningLab/DeepInterAware README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"9f1cef47f48b4b3b0084e6de73d271166f5c85344603b6aac5a64521eac10a9a","artifact_url":"https://raw.githubusercontent.com/BioMedicalBigDataMiningLab/DeepInterAware/11a283264fb7177f842d56d5c6b49f9bb7f10abd/README.md","retrieved_at":"2026-09-16T20:42:56.583634+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/BioMedicalBigDataMiningLab/DeepInterAware/blob/11a283264fb7177f842d56d5c6b49f9bb7f10abd/README.md","version":"11a283264fb7177f842d56d5c6b49f9bb7f10abd"}} {"id":"evidence-reported-detire-viral-metagenomes-2023-readme-md","kind":"source","name":"crazyinter/DETIRE README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"cb1ac5f1df284b248f02fd1ac8f6e43a825c7c918e64cba381df3792e2403e9f","artifact_url":"https://raw.githubusercontent.com/crazyinter/DETIRE/6b48c5bcb1303abe593173633d1f13da1d8d5869/README.md","retrieved_at":"2026-09-16T19:54:13.776207+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/crazyinter/DETIRE/blob/6b48c5bcb1303abe593173633d1f13da1d8d5869/README.md","version":"6b48c5bcb1303abe593173633d1f13da1d8d5869"}} {"id":"evidence-reported-dna-foundation-models-2025-readme-md","kind":"source","name":"ChongWuLab/dna_foundation_benchmark README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"e678c69a91fd0b709864d2c0b83f33e273b82e389dd07b6f380b76236631fac2","artifact_url":"https://raw.githubusercontent.com/ChongWuLab/dna_foundation_benchmark/3f4c81ce066f3c47422a83466b085aac1a6be902/README.md","retrieved_at":"2026-09-16T19:54:14.870570+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/ChongWuLab/dna_foundation_benchmark/blob/3f4c81ce066f3c47422a83466b085aac1a6be902/README.md","version":"3f4c81ce066f3c47422a83466b085aac1a6be902"}} {"id":"evidence-reported-enhancer-position-encoding-2024-readme-md","kind":"source","name":"xing1999/PDCNN README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"77382e4a76f669866bb8e81551aba59b5993a1ff816c9b530a823940e6db1992","artifact_url":"https://raw.githubusercontent.com/xing1999/PDCNN/ff302344eb1a03bca4802408d03651cb4d2abc99/README.md","retrieved_at":"2026-09-16T19:54:14.982961+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/xing1999/PDCNN/blob/ff302344eb1a03bca4802408d03651cb4d2abc99/README.md","version":"ff302344eb1a03bca4802408d03651cb4d2abc99"}} {"id":"evidence-reported-ernie-rna-2025-license","kind":"source","name":"Bruce-ywj/ERNIE-RNA LICENSE","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"b7aa7b89bff7b3c4cca2f78dc0900f038a7253bcbea9b780eca30b2bc1610ffb","artifact_url":"https://raw.githubusercontent.com/Bruce-ywj/ERNIE-RNA/43bc06de1088ed03ffd7de918ad4b2c2a3346a43/LICENSE","retrieved_at":"2026-09-16T19:54:15.083678+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/Bruce-ywj/ERNIE-RNA/blob/43bc06de1088ed03ffd7de918ad4b2c2a3346a43/LICENSE","version":"43bc06de1088ed03ffd7de918ad4b2c2a3346a43"}} {"id":"evidence-reported-ernie-rna-2025-readme-md","kind":"source","name":"Bruce-ywj/ERNIE-RNA README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"d2b2ca2f33c4d4d1dac63659cff494631732b7ec87ac07ad99a0826e55bfa603","artifact_url":"https://raw.githubusercontent.com/Bruce-ywj/ERNIE-RNA/43bc06de1088ed03ffd7de918ad4b2c2a3346a43/README.md","retrieved_at":"2026-09-16T19:54:15.083678+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/Bruce-ywj/ERNIE-RNA/blob/43bc06de1088ed03ffd7de918ad4b2c2a3346a43/README.md","version":"43bc06de1088ed03ffd7de918ad4b2c2a3346a43"}} {"id":"evidence-reported-fujisan-2024-readme-md","kind":"source","name":"sfujita0601/FUJISAN README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"cb023829651bce1023dd11f603db6731efbff534e04e863821b8677fcec5a19e","artifact_url":"https://raw.githubusercontent.com/sfujita0601/FUJISAN/588daf65810c49b021af9e47b11ec752986dcb94/README.md","retrieved_at":"2026-09-16T19:54:15.262300+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/sfujita0601/FUJISAN/blob/588daf65810c49b021af9e47b11ec752986dcb94/README.md","version":"588daf65810c49b021af9e47b11ec752986dcb94"}} {"id":"evidence-reported-fusion-breakpoint-foundation-models-2026-license","kind":"source","name":"kbi-fbmi/articles--2026fusionEmbBenchmark LICENSE","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"fbc0aec4a1e90f07e8867eda6bc64940755f69d49c11ce284c6b6075df98eabf","artifact_url":"https://raw.githubusercontent.com/kbi-fbmi/articles--2026fusionEmbBenchmark/085a6d7d2f899b0f62d764f35d1248b2eda567da/LICENSE","retrieved_at":"2026-09-16T19:54:16.293121+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/kbi-fbmi/articles--2026fusionEmbBenchmark/blob/085a6d7d2f899b0f62d764f35d1248b2eda567da/LICENSE","version":"085a6d7d2f899b0f62d764f35d1248b2eda567da"}} {"id":"evidence-reported-fusion-breakpoint-foundation-models-2026-readme-md","kind":"source","name":"kbi-fbmi/articles--2026fusionEmbBenchmark README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"895b3a5d9f861be27e7f19b77b5f875f72abcea680ed444d8794d0d6604b4dc3","artifact_url":"https://raw.githubusercontent.com/kbi-fbmi/articles--2026fusionEmbBenchmark/085a6d7d2f899b0f62d764f35d1248b2eda567da/README.md","retrieved_at":"2026-09-16T19:54:16.293121+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/kbi-fbmi/articles--2026fusionEmbBenchmark/blob/085a6d7d2f899b0f62d764f35d1248b2eda567da/README.md","version":"085a6d7d2f899b0f62d764f35d1248b2eda567da"}} {"id":"evidence-reported-genept-2024-readme-md","kind":"source","name":"yiqunchen/GenePT README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"8740cf98377f964131399de818db5de938db4d9e6abe20fac3cd0354887a99c2","artifact_url":"https://raw.githubusercontent.com/yiqunchen/GenePT/3602699e7425a7be577771f8f07e218db6c79b9f/README.md","retrieved_at":"2026-09-16T19:54:16.422790+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/yiqunchen/GenePT/blob/3602699e7425a7be577771f8f07e218db6c79b9f/README.md","version":"3602699e7425a7be577771f8f07e218db6c79b9f"}} {"id":"evidence-reported-genomeocean-2025-license","kind":"source","name":"jgi-genomeocean/genomeocean LICENSE","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"1700c41a252e2fa9dd284a70387de610f2b47e663b8954c0538f1a9b9227e7a9","artifact_url":"https://raw.githubusercontent.com/jgi-genomeocean/genomeocean/06fa433169539a3c84d7366a663933b888a5386b/LICENSE","retrieved_at":"2026-09-16T19:54:16.470820+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/jgi-genomeocean/genomeocean/blob/06fa433169539a3c84d7366a663933b888a5386b/LICENSE","version":"06fa433169539a3c84d7366a663933b888a5386b"}} {"id":"evidence-reported-genomeocean-2025-readme-md","kind":"source","name":"jgi-genomeocean/genomeocean README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"57bfdecac0eee10307ebc396fed2f9398492ac9d493eccb86e3607599f8ae907","artifact_url":"https://raw.githubusercontent.com/jgi-genomeocean/genomeocean/06fa433169539a3c84d7366a663933b888a5386b/README.md","retrieved_at":"2026-09-16T19:54:16.470820+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/jgi-genomeocean/genomeocean/blob/06fa433169539a3c84d7366a663933b888a5386b/README.md","version":"06fa433169539a3c84d7366a663933b888a5386b"}} {"id":"evidence-reported-gremln-2026-license-md","kind":"source","name":"czi-ai/GREmLN LICENSE.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"4538ca25a86f413b4f0856752768517f93a98d76f53ee09989d9b162144a8f6d","artifact_url":"https://raw.githubusercontent.com/czi-ai/GREmLN/e1c5d8edbe2fe96568ff5451f15bc691722bb6a1/LICENSE.md","retrieved_at":"2026-09-16T20:42:56.518821+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/czi-ai/GREmLN/blob/e1c5d8edbe2fe96568ff5451f15bc691722bb6a1/LICENSE.md","version":"e1c5d8edbe2fe96568ff5451f15bc691722bb6a1"}} {"id":"evidence-reported-gremln-2026-readme-md","kind":"source","name":"czi-ai/GREmLN README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"e6365c6df93ad707c57e04d57b6e3aa4121a1d36672e9a4f75db4a8ac664a858","artifact_url":"https://raw.githubusercontent.com/czi-ai/GREmLN/e1c5d8edbe2fe96568ff5451f15bc691722bb6a1/README.md","retrieved_at":"2026-09-16T20:42:56.518821+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/czi-ai/GREmLN/blob/e1c5d8edbe2fe96568ff5451f15bc691722bb6a1/README.md","version":"e1c5d8edbe2fe96568ff5451f15bc691722bb6a1"}} {"id":"evidence-reported-gsmformer-ppi-2026-readme-md","kind":"source","name":"ChervovNikita/gsmformer-ppi README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"45003a18fbc137225141cc473b1399c47952183bf784cef8904d8ec3e298cf6b","artifact_url":"https://raw.githubusercontent.com/ChervovNikita/gsmformer-ppi/db9886e8b295f35b544a5703659e2a9115ce9e22/README.md","retrieved_at":"2026-09-16T19:54:16.546691+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/ChervovNikita/gsmformer-ppi/blob/db9886e8b295f35b544a5703659e2a9115ce9e22/README.md","version":"db9886e8b295f35b544a5703659e2a9115ce9e22"}} {"id":"evidence-reported-hi-enhancer-2025-readme-txt","kind":"source","name":"emanlee/Hi-Enhancer README.txt","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"d2880c9ae82730700607d25c7347641dba5484604e841a696be874589d744942","artifact_url":"https://raw.githubusercontent.com/emanlee/Hi-Enhancer/435bb1cc9ec2909d6ee551bb53c56f6a7bdde8fd/README.txt","retrieved_at":"2026-09-16T19:54:16.659616+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/emanlee/Hi-Enhancer/blob/435bb1cc9ec2909d6ee551bb53c56f6a7bdde8fd/README.txt","version":"435bb1cc9ec2909d6ee551bb53c56f6a7bdde8fd"}} {"id":"evidence-reported-ibex-2025-license","kind":"source","name":"prescient-design/ibex LICENSE","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"c71d239df91726fc519c6eb72d318ec65820627232b2f796219e87dcf35d0ab4","artifact_url":"https://raw.githubusercontent.com/prescient-design/ibex/2e785563806a3b63600eeaa2107d1254ddf5d196/LICENSE","retrieved_at":"2026-09-16T19:54:16.662235+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/prescient-design/ibex/blob/2e785563806a3b63600eeaa2107d1254ddf5d196/LICENSE","version":"2e785563806a3b63600eeaa2107d1254ddf5d196"}} {"id":"evidence-reported-ibex-2025-readme-md","kind":"source","name":"prescient-design/ibex README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"e69723730602ff47d4f33e2eb37140bcfeff8c8a491ea15580f973fa51d1b4a9","artifact_url":"https://raw.githubusercontent.com/prescient-design/ibex/2e785563806a3b63600eeaa2107d1254ddf5d196/README.md","retrieved_at":"2026-09-16T19:54:16.662235+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/prescient-design/ibex/blob/2e785563806a3b63600eeaa2107d1254ddf5d196/README.md","version":"2e785563806a3b63600eeaa2107d1254ddf5d196"}} {"id":"evidence-reported-icctax-2025-readme-md","kind":"source","name":"Ying-Lab/ICCTax README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"749dcacc87c288d7a248a9bd5f58d3ebe39f229ff8853b3fff694957541e0668","artifact_url":"https://raw.githubusercontent.com/Ying-Lab/ICCTax/6b7381c7111bde6d40324cd033501204bf3ac3bc/README.md","retrieved_at":"2026-09-16T19:54:16.993236+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/Ying-Lab/ICCTax/blob/6b7381c7111bde6d40324cd033501204bf3ac3bc/README.md","version":"6b7381c7111bde6d40324cd033501204bf3ac3bc"}} {"id":"evidence-reported-insilico-perturbation-auprc-2025-readme-md","kind":"source","name":"hxzhu491/Cell-Perturbation-evaluation-Metric README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"f72281cb1777ca70ba37e974ad72ed7fcfd83639e7e008b0e83824a02208c567","artifact_url":"https://raw.githubusercontent.com/hxzhu491/Cell-Perturbation-evaluation-Metric/3b5f8a2ed001c074936287ece478747376c8f5bf/README.md","retrieved_at":"2026-09-16T19:54:17.543702+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/hxzhu491/Cell-Perturbation-evaluation-Metric/blob/3b5f8a2ed001c074936287ece478747376c8f5bf/README.md","version":"3b5f8a2ed001c074936287ece478747376c8f5bf"}} {"id":"evidence-reported-ipro70-original","kind":"source","name":"iPro70-FMWin original paper","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"e4fa53c5574961acf4a93bf251b71c248025c64b1b79e74da40ff73091d8b205","artifact_url":"https://rafsanjani.pythonanywhere.com/static/Papers/iPro70.pdf","retrieved_at":"2026-09-16T20:40:28.842295+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://rafsanjani.pythonanywhere.com/static/Papers/iPro70.pdf","version":"10.1007/s00438-018-1487-5"}} {"id":"evidence-reported-ipromp-2025-readme-md","kind":"source","name":"Jackie-Suv/iPro-MP README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"8245ed3d5199578dc5c47af893eb96b28b100ac5fc88dfc2585074096d7f8191","artifact_url":"https://raw.githubusercontent.com/Jackie-Suv/iPro-MP/4266b521bc6617db939c5871cb1b6850dff63fdb/README.md","retrieved_at":"2026-09-16T19:54:17.673506+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/Jackie-Suv/iPro-MP/blob/4266b521bc6617db939c5871cb1b6850dff63fdb/README.md","version":"4266b521bc6617db939c5871cb1b6850dff63fdb"}} {"id":"evidence-reported-kmetashot-2025-license","kind":"source","name":"gdefazio/kMetaShot LICENSE","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"9e54fb30a82e1879c56cb0206a8107659190dd76396029df8b7c5363f7757cfc","artifact_url":"https://raw.githubusercontent.com/gdefazio/kMetaShot/95dac648aba94d119d20929478f6c3955206f9d4/LICENSE","retrieved_at":"2026-09-16T19:54:17.812170+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/gdefazio/kMetaShot/blob/95dac648aba94d119d20929478f6c3955206f9d4/LICENSE","version":"95dac648aba94d119d20929478f6c3955206f9d4"}} {"id":"evidence-reported-kmetashot-2025-readme-md","kind":"source","name":"gdefazio/kMetaShot README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"0b1a56d4d6106e664ffdbfa0771cb6d9be65b55b1ada5c6b5299188dc00fc7dc","artifact_url":"https://raw.githubusercontent.com/gdefazio/kMetaShot/95dac648aba94d119d20929478f6c3955206f9d4/README.md","retrieved_at":"2026-09-16T19:54:17.812170+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/gdefazio/kMetaShot/blob/95dac648aba94d119d20929478f6c3955206f9d4/README.md","version":"95dac648aba94d119d20929478f6c3955206f9d4"}} {"id":"evidence-reported-lemur-magnet-2024-license","kind":"source","name":"treangenlab/lemur LICENSE","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"493e198a2a2b2ddf102c371bdd51bf2be1e974e30117321d1a8a50e531b81787","artifact_url":"https://raw.githubusercontent.com/treangenlab/lemur/eda2cb57727b72fc5b1fb28be1fe45a4826100f9/LICENSE","retrieved_at":"2026-09-16T19:54:18.142852+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/treangenlab/lemur/blob/eda2cb57727b72fc5b1fb28be1fe45a4826100f9/LICENSE","version":"eda2cb57727b72fc5b1fb28be1fe45a4826100f9"}} {"id":"evidence-reported-lemur-magnet-2024-readme-md","kind":"source","name":"treangenlab/lemur README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"cbe1079a582be626c015d00a80dffb199090761ed5c7d621def0aef28d032aeb","artifact_url":"https://raw.githubusercontent.com/treangenlab/lemur/eda2cb57727b72fc5b1fb28be1fe45a4826100f9/README.md","retrieved_at":"2026-09-16T19:54:18.142852+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/treangenlab/lemur/blob/eda2cb57727b72fc5b1fb28be1fe45a4826100f9/README.md","version":"eda2cb57727b72fc5b1fb28be1fe45a4826100f9"}} {"id":"evidence-reported-ligand-affinity-meta-model-2024-license","kind":"source","name":"Lee1701/Lee2023a LICENSE","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"3972dc9744f6499f0f9b2dbf76696f2ae7ad8af9b23dde66d6af86c9dfb36986","artifact_url":"https://raw.githubusercontent.com/Lee1701/Lee2023a/92def517a0edcb0470826353afd81899bcfdb4b1/LICENSE","retrieved_at":"2026-09-16T20:30:16.574955+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/Lee1701/Lee2023a/blob/92def517a0edcb0470826353afd81899bcfdb4b1/LICENSE","version":"92def517a0edcb0470826353afd81899bcfdb4b1"}} {"id":"evidence-reported-ligand-affinity-meta-model-2024-readme-md","kind":"source","name":"Lee1701/Lee2023a README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"a428cb0e912f6608de6a17ec162193eaefa9f96074278ac9d9b2ecf051cdbedd","artifact_url":"https://raw.githubusercontent.com/Lee1701/Lee2023a/92def517a0edcb0470826353afd81899bcfdb4b1/README.md","retrieved_at":"2026-09-16T20:30:16.574955+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/Lee1701/Lee2023a/blob/92def517a0edcb0470826353afd81899bcfdb4b1/README.md","version":"92def517a0edcb0470826353afd81899bcfdb4b1"}} {"id":"evidence-reported-mdl4microbiome-2022-readme-md","kind":"source","name":"DMnBI/MDL4Microbiome README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"8737382048b17436607ae12302dfcfed021efcc39fa849d8936c6b9ac9f52d49","artifact_url":"https://raw.githubusercontent.com/DMnBI/MDL4Microbiome/0b2076cbd31af62224bd7a90ef0130e4cac50017/README.md","retrieved_at":"2026-09-16T19:54:18.201417+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/DMnBI/MDL4Microbiome/blob/0b2076cbd31af62224bd7a90ef0130e4cac50017/README.md","version":"0b2076cbd31af62224bd7a90ef0130e4cac50017"}} {"id":"evidence-reported-megsite-2025-readme-md","kind":"source","name":"pengsl-lab/MegSite README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"226fee71219189aded7aa165f7de40c85d2f96f4106089a8cd047793b524d18c","artifact_url":"https://raw.githubusercontent.com/pengsl-lab/MegSite/4d1f5441f15bd20eb3e7c9e5be2d4ca5a857f46b/README.md","retrieved_at":"2026-09-16T19:54:18.210404+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/pengsl-lab/MegSite/blob/4d1f5441f15bd20eb3e7c9e5be2d4ca5a857f46b/README.md","version":"4d1f5441f15bd20eb3e7c9e5be2d4ca5a857f46b"}} {"id":"evidence-reported-molas-2026-license","kind":"source","name":"BradWangW/MolAS LICENSE","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"33f037265e2bad90af34079cbee68538db96193a25db41d0e227b085fcb6177c","artifact_url":"https://raw.githubusercontent.com/BradWangW/MolAS/a6c417216ddb992f7dc513d511d0429aece4bd61/LICENSE","retrieved_at":"2026-09-16T20:36:11.670546+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/BradWangW/MolAS/blob/a6c417216ddb992f7dc513d511d0429aece4bd61/LICENSE","version":"a6c417216ddb992f7dc513d511d0429aece4bd61"}} {"id":"evidence-reported-molas-2026-readme-md","kind":"source","name":"BradWangW/MolAS README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"b7a057042375a8c796278242b1b89d8ff7a1cd99203ff2fe31e7d1df9226256e","artifact_url":"https://raw.githubusercontent.com/BradWangW/MolAS/a6c417216ddb992f7dc513d511d0429aece4bd61/README.md","retrieved_at":"2026-09-16T20:36:11.670546+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/BradWangW/MolAS/blob/a6c417216ddb992f7dc513d511d0429aece4bd61/README.md","version":"a6c417216ddb992f7dc513d511d0429aece4bd61"}} {"id":"evidence-reported-mouse-geneformer-2025-readme-md","kind":"source","name":"machine-perception-robotics-group/Mouse-Geneformer README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"1eefd8677ff2644f55272a1412cf612e1264f92456ae3b701598d9a5b558bc4f","artifact_url":"https://raw.githubusercontent.com/machine-perception-robotics-group/Mouse-Geneformer/ed17d455193ed4c4d93230a8f5f2de349cf20c81/README.md","retrieved_at":"2026-09-16T19:54:18.217286+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/machine-perception-robotics-group/Mouse-Geneformer/blob/ed17d455193ed4c4d93230a8f5f2de349cf20c81/README.md","version":"ed17d455193ed4c4d93230a8f5f2de349cf20c81"}} {"id":"evidence-reported-mrna-lm-2025-license-txt","kind":"source","name":"Sanofi-Public/mRNA-LM LICENSE.txt","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"6503f6c09579df890e003db39c9b7c299592d25eee97066241b0e7ac9212a1ea","artifact_url":"https://raw.githubusercontent.com/Sanofi-Public/mRNA-LM/d7538c9aadbceb59a8832904292b279d0a4c2d12/LICENSE.txt","retrieved_at":"2026-09-16T19:54:18.716122+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/Sanofi-Public/mRNA-LM/blob/d7538c9aadbceb59a8832904292b279d0a4c2d12/LICENSE.txt","version":"d7538c9aadbceb59a8832904292b279d0a4c2d12"}} {"id":"evidence-reported-mrna-lm-2025-readme-md","kind":"source","name":"Sanofi-Public/mRNA-LM README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"896d2f64f8a236c9f99cad115feb97dab369bbf4c42bb6025df01bbae0bd740b","artifact_url":"https://raw.githubusercontent.com/Sanofi-Public/mRNA-LM/d7538c9aadbceb59a8832904292b279d0a4c2d12/README.md","retrieved_at":"2026-09-16T19:54:18.716122+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/Sanofi-Public/mRNA-LM/blob/d7538c9aadbceb59a8832904292b279d0a4c2d12/README.md","version":"d7538c9aadbceb59a8832904292b279d0a4c2d12"}} {"id":"evidence-reported-mrna-protein-diversity-2026-readme-md","kind":"source","name":"cobisLab/mRPI-issue README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"10b6607fc40d285e20b681a7584b3e980265d15dcc79c3ac073aa4f030cfc141","artifact_url":"https://raw.githubusercontent.com/cobisLab/mRPI-issue/0f2d27666c4876f00d4c4e6cb3bec0e1629d214c/README.md","retrieved_at":"2026-09-16T19:54:19.069409+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/cobisLab/mRPI-issue/blob/0f2d27666c4876f00d4c4e6cb3bec0e1629d214c/README.md","version":"0f2d27666c4876f00d4c4e6cb3bec0e1629d214c"}} {"id":"evidence-reported-mrnabert-2025-license","kind":"source","name":"yyly6/mRNABERT LICENSE","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"c71d239df91726fc519c6eb72d318ec65820627232b2f796219e87dcf35d0ab4","artifact_url":"https://raw.githubusercontent.com/yyly6/mRNABERT/893ccc920bb9be02a4677d14d96b03126da17689/LICENSE","retrieved_at":"2026-09-16T19:54:19.395631+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/yyly6/mRNABERT/blob/893ccc920bb9be02a4677d14d96b03126da17689/LICENSE","version":"893ccc920bb9be02a4677d14d96b03126da17689"}} {"id":"evidence-reported-mrnabert-2025-readme-md","kind":"source","name":"yyly6/mRNABERT README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"bd50f1d7a71b1fd265b8b4c5590358d9bdabc4b6d897cd3b2ec6ba7ad5f7d2ff","artifact_url":"https://raw.githubusercontent.com/yyly6/mRNABERT/893ccc920bb9be02a4677d14d96b03126da17689/README.md","retrieved_at":"2026-09-16T19:54:19.395631+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/yyly6/mRNABERT/blob/893ccc920bb9be02a4677d14d96b03126da17689/README.md","version":"893ccc920bb9be02a4677d14d96b03126da17689"}} {"id":"evidence-reported-nabas-plus-2025-license","kind":"source","name":"TakacsBertalan/NABAS_paper_scripts LICENSE","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"c84442114f4da9b2593efa430845e11a0605b0c16259feb906661bc0bd504fb8","artifact_url":"https://raw.githubusercontent.com/TakacsBertalan/NABAS_paper_scripts/7cab4d317a2c362988e7b96fb33f92a9c79a9fdc/LICENSE","retrieved_at":"2026-09-16T19:54:19.540591+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/TakacsBertalan/NABAS_paper_scripts/blob/7cab4d317a2c362988e7b96fb33f92a9c79a9fdc/LICENSE","version":"7cab4d317a2c362988e7b96fb33f92a9c79a9fdc"}} {"id":"evidence-reported-nabas-plus-2025-readme-md","kind":"source","name":"TakacsBertalan/NABAS_paper_scripts README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"4dd306069e898af754cdf810adbcf7542a24918563d610933f50f48d41aeecac","artifact_url":"https://raw.githubusercontent.com/TakacsBertalan/NABAS_paper_scripts/7cab4d317a2c362988e7b96fb33f92a9c79a9fdc/README.md","retrieved_at":"2026-09-16T19:54:19.540591+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/TakacsBertalan/NABAS_paper_scripts/blob/7cab4d317a2c362988e7b96fb33f92a9c79a9fdc/README.md","version":"7cab4d317a2c362988e7b96fb33f92a9c79a9fdc"}} {"id":"evidence-reported-ncd-metagenomics-2026-license","kind":"source","name":"ghproducts/genomics-ncd-gzip LICENSE","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"a2010f343487d3f7618affe54f789f5487602331c0a8d03f49e9a7c547cf0499","artifact_url":"https://raw.githubusercontent.com/ghproducts/genomics-ncd-gzip/d5bb37be194a2716bc65543c65a26a65eebc1849/LICENSE","retrieved_at":"2026-09-16T19:54:19.565893+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/ghproducts/genomics-ncd-gzip/blob/d5bb37be194a2716bc65543c65a26a65eebc1849/LICENSE","version":"d5bb37be194a2716bc65543c65a26a65eebc1849"}} {"id":"evidence-reported-ncd-metagenomics-2026-readme-md","kind":"source","name":"ghproducts/genomics-ncd-gzip README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"79b4134061f21aa3baed67fa656f055686db77adb08ac0ebe2dd6c8a5d4fb92b","artifact_url":"https://raw.githubusercontent.com/ghproducts/genomics-ncd-gzip/d5bb37be194a2716bc65543c65a26a65eebc1849/README.md","retrieved_at":"2026-09-16T19:54:19.565893+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/ghproducts/genomics-ncd-gzip/blob/d5bb37be194a2716bc65543c65a26a65eebc1849/README.md","version":"d5bb37be194a2716bc65543c65a26a65eebc1849"}} {"id":"evidence-reported-pc-mer-2024-readme-md","kind":"source","name":"SAkbari93/PC-mer_Metagenomics README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"07d86adec2ed17ed488c5af3d6348e9da967c6b6804d57093f2a8576abf132ed","artifact_url":"https://raw.githubusercontent.com/SAkbari93/PC-mer_Metagenomics/5c5f89dcaec5098372ad1fe82d4215186fe417c1/README.md","retrieved_at":"2026-09-16T19:54:20.219994+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/SAkbari93/PC-mer_Metagenomics/blob/5c5f89dcaec5098372ad1fe82d4215186fe417c1/README.md","version":"5c5f89dcaec5098372ad1fe82d4215186fe417c1"}} {"id":"evidence-reported-phylogpn-2025-license","kind":"source","name":"songlab-cal/gpn LICENSE","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"cef312cb27bc4aae45e0ca7130643ca36f9aa4165ef09bcddf5ba5eb6fb76e90","artifact_url":"https://raw.githubusercontent.com/songlab-cal/gpn/6f28c81bcbfe7d65cb6d8ece9ce88f87ca583791/LICENSE","retrieved_at":"2026-09-16T19:54:20.313775+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/songlab-cal/gpn/blob/6f28c81bcbfe7d65cb6d8ece9ce88f87ca583791/LICENSE","version":"6f28c81bcbfe7d65cb6d8ece9ce88f87ca583791"}} {"id":"evidence-reported-phylogpn-2025-readme-md","kind":"source","name":"songlab-cal/gpn README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"0312fb9646aef630a30cd3039337b31a48f4345f4dd1cdd71edcc483afd8b9d0","artifact_url":"https://raw.githubusercontent.com/songlab-cal/gpn/6f28c81bcbfe7d65cb6d8ece9ce88f87ca583791/README.md","retrieved_at":"2026-09-16T19:54:20.313775+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/songlab-cal/gpn/blob/6f28c81bcbfe7d65cb6d8ece9ce88f87ca583791/README.md","version":"6f28c81bcbfe7d65cb6d8ece9ce88f87ca583791"}} {"id":"evidence-reported-plantcad2-2025-license","kind":"source","name":"plantcad/plantcad LICENSE","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"c71d239df91726fc519c6eb72d318ec65820627232b2f796219e87dcf35d0ab4","artifact_url":"https://raw.githubusercontent.com/plantcad/plantcad/7240f0238f869b3ac25e4b5ad0996fad96ede9db/LICENSE","retrieved_at":"2026-09-16T19:54:20.832487+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/plantcad/plantcad/blob/7240f0238f869b3ac25e4b5ad0996fad96ede9db/LICENSE","version":"7240f0238f869b3ac25e4b5ad0996fad96ede9db"}} {"id":"evidence-reported-plantcad2-2025-readme-md","kind":"source","name":"plantcad/plantcad README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"74ef9550db6752e6d99441c48751919e172ce5e2ce83cf57deaa08f24df1df8e","artifact_url":"https://raw.githubusercontent.com/plantcad/plantcad/7240f0238f869b3ac25e4b5ad0996fad96ede9db/README.md","retrieved_at":"2026-09-16T19:54:20.832487+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/plantcad/plantcad/blob/7240f0238f869b3ac25e4b5ad0996fad96ede9db/README.md","version":"7240f0238f869b3ac25e4b5ad0996fad96ede9db"}} {"id":"evidence-reported-prime-2026-license-md","kind":"source","name":"lanl/prime LICENSE.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"1a852d32b1c472babd61e142e4c3674ac971112bd438a8f3eb006af752390c9a","artifact_url":"https://raw.githubusercontent.com/lanl/prime/d940c51aa0f475b8945789e89761fab0687d5b74/LICENSE.md","retrieved_at":"2026-09-16T19:54:21.047073+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/lanl/prime/blob/d940c51aa0f475b8945789e89761fab0687d5b74/LICENSE.md","version":"d940c51aa0f475b8945789e89761fab0687d5b74"}} {"id":"evidence-reported-prime-2026-readme-md","kind":"source","name":"lanl/prime README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"35a8c35aad87c03f8a5afcb66dd1b6477ff7ab187e77d5951503773d7dfbc896","artifact_url":"https://raw.githubusercontent.com/lanl/prime/d940c51aa0f475b8945789e89761fab0687d5b74/README.md","retrieved_at":"2026-09-16T19:54:21.047073+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/lanl/prime/blob/d940c51aa0f475b8945789e89761fab0687d5b74/README.md","version":"d940c51aa0f475b8945789e89761fab0687d5b74"}} {"id":"evidence-reported-prokbert-2024-license","kind":"source","name":"nbrg-ppcu/prokbert LICENSE","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"66e17f3a57ace034ab422667c02d75134bb531619d9491f3abc7ea42bb8a6643","artifact_url":"https://raw.githubusercontent.com/nbrg-ppcu/prokbert/8670ae92b816cff158a0b85647a8dea122e251eb/LICENSE","retrieved_at":"2026-09-16T19:54:21.078483+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/nbrg-ppcu/prokbert/blob/8670ae92b816cff158a0b85647a8dea122e251eb/LICENSE","version":"8670ae92b816cff158a0b85647a8dea122e251eb"}} {"id":"evidence-reported-prokbert-2024-readme-md","kind":"source","name":"nbrg-ppcu/prokbert README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"29de39c6ad006ce704ab14240cfd97af93da411fb63ec89ebe499c2646928cfc","artifact_url":"https://raw.githubusercontent.com/nbrg-ppcu/prokbert/8670ae92b816cff158a0b85647a8dea122e251eb/README.md","retrieved_at":"2026-09-16T19:54:21.078483+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/nbrg-ppcu/prokbert/blob/8670ae92b816cff158a0b85647a8dea122e251eb/README.md","version":"8670ae92b816cff158a0b85647a8dea122e251eb"}} {"id":"evidence-reported-protein-binding-sites-2023-readme-md","kind":"source","name":"houzl3416/EDLMPPI README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"95a7d8240f3e685689d2f978e7081a0ff7dd8d058fe850f54662f87530632e83","artifact_url":"https://raw.githubusercontent.com/houzl3416/EDLMPPI/78e4a7b36bb83ccf4274786b859125178804f434/README.md","retrieved_at":"2026-09-16T19:54:21.260563+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/houzl3416/EDLMPPI/blob/78e4a7b36bb83ccf4274786b859125178804f434/README.md","version":"78e4a7b36bb83ccf4274786b859125178804f434"}} {"id":"evidence-reported-pst-2025-license","kind":"source","name":"BorgwardtLab/PST LICENSE","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"f2ce3382be97305fc31c8b061f1f08969e6f95c677db5b8a7c3bc81352237dfb","artifact_url":"https://raw.githubusercontent.com/BorgwardtLab/PST/57d9dcd8200900504a19de459450e137867262d7/LICENSE","retrieved_at":"2026-09-16T19:54:21.541747+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/BorgwardtLab/PST/blob/57d9dcd8200900504a19de459450e137867262d7/LICENSE","version":"57d9dcd8200900504a19de459450e137867262d7"}} {"id":"evidence-reported-pst-2025-readme-md","kind":"source","name":"BorgwardtLab/PST README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"e3422ce3da04d004549e37cfcbe9a509d53a8cd9217b39598ac10b6b444a8121","artifact_url":"https://raw.githubusercontent.com/BorgwardtLab/PST/57d9dcd8200900504a19de459450e137867262d7/README.md","retrieved_at":"2026-09-16T19:54:21.541747+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/BorgwardtLab/PST/blob/57d9dcd8200900504a19de459450e137867262d7/README.md","version":"57d9dcd8200900504a19de459450e137867262d7"}} {"id":"evidence-reported-r3design-2025-readme-md","kind":"source","name":"A4Bio/R3Design readme.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"132adf2bb9ae5cc090a33a01f2534bf37a5a7daee3b4a4f19b28803d0655da79","artifact_url":"https://raw.githubusercontent.com/A4Bio/R3Design/c05dc4b35100949011c77c08f61a9e80280987ce/readme.md","retrieved_at":"2026-09-16T19:54:22.378999+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/A4Bio/R3Design/blob/c05dc4b35100949011c77c08f61a9e80280987ce/readme.md","version":"c05dc4b35100949011c77c08f61a9e80280987ce"}} {"id":"evidence-reported-rewire-license","kind":"source","name":"Rewire benchmark runner licence","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"4fff574dd02d01f7cba42c9a4472bce39bb96f9b650a65f51958ab79bc829ea4","artifact_url":"https://github.com/rewire-bio/rewire-benchmarks/blob/bee9133b83f3aedaf2bbb9013f1875515845607e/LICENSE","retrieved_at":"2026-09-16T20:47:15.959417+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/rewire-bio/rewire-benchmarks/blob/bee9133b83f3aedaf2bbb9013f1875515845607e/LICENSE","version":"bee9133b83f3aedaf2bbb9013f1875515845607e"}} {"id":"evidence-reported-rewire-run-baseline","kind":"source","name":"MFASS v2 run_baseline.py","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"7066ae357aac960ddd6e95ee55a98dcd876b619edfadae1e1d9d03eb3f5b0acc","artifact_url":"https://github.com/rewire-bio/rewire-benchmarks/blob/bee9133b83f3aedaf2bbb9013f1875515845607e/benchmarks/mfass/src/mfass/run_baseline.py","retrieved_at":"2026-09-16T20:47:15.959417+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/rewire-bio/rewire-benchmarks/blob/bee9133b83f3aedaf2bbb9013f1875515845607e/benchmarks/mfass/src/mfass/run_baseline.py","version":"bee9133b83f3aedaf2bbb9013f1875515845607e"}} {"id":"evidence-reported-rewire-run-dnabert2","kind":"source","name":"MFASS v2 run_dnabert2.py","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"b7ebb4324e22cb421482181c52722c64cac5fec7c43161c082fc34e9920c4a85","artifact_url":"https://github.com/rewire-bio/rewire-benchmarks/blob/bee9133b83f3aedaf2bbb9013f1875515845607e/benchmarks/mfass/src/mfass/run_dnabert2.py","retrieved_at":"2026-09-16T20:47:15.959417+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/rewire-bio/rewire-benchmarks/blob/bee9133b83f3aedaf2bbb9013f1875515845607e/benchmarks/mfass/src/mfass/run_dnabert2.py","version":"bee9133b83f3aedaf2bbb9013f1875515845607e"}} {"id":"evidence-reported-rewire-run-pangolin","kind":"source","name":"MFASS v2 run_pangolin.py","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"83e5b71d7265c63fb0874d7b493ac2c8d205f645c079b09f350d69b44e5c921b","artifact_url":"https://github.com/rewire-bio/rewire-benchmarks/blob/bee9133b83f3aedaf2bbb9013f1875515845607e/benchmarks/mfass/src/mfass/run_pangolin.py","retrieved_at":"2026-09-16T20:47:15.959417+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/rewire-bio/rewire-benchmarks/blob/bee9133b83f3aedaf2bbb9013f1875515845607e/benchmarks/mfass/src/mfass/run_pangolin.py","version":"bee9133b83f3aedaf2bbb9013f1875515845607e"}} {"id":"evidence-reported-rewire-run-spliceai","kind":"source","name":"MFASS v2 run_spliceai.py","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"3709bc12997744b77ce2f9ea4de79cdcafa6ffc136b61bb020f57e43acbe8685","artifact_url":"https://github.com/rewire-bio/rewire-benchmarks/blob/bee9133b83f3aedaf2bbb9013f1875515845607e/benchmarks/mfass/src/mfass/run_spliceai.py","retrieved_at":"2026-09-16T20:47:15.959417+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/rewire-bio/rewire-benchmarks/blob/bee9133b83f3aedaf2bbb9013f1875515845607e/benchmarks/mfass/src/mfass/run_spliceai.py","version":"bee9133b83f3aedaf2bbb9013f1875515845607e"}} {"id":"evidence-reported-rlsite-rna-binding-2025-readme-md","kind":"source","name":"SaisaiSun/RLsite README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"56a8c16f47bec013c53fb2ce7ecc1c6f87db0b724176677b00d44279b4b3ec09","artifact_url":"https://raw.githubusercontent.com/SaisaiSun/RLsite/3d4a294aab295cbc1afd56b9ea6ced96e0717304/README.md","retrieved_at":"2026-09-16T19:54:22.406853+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/SaisaiSun/RLsite/blob/3d4a294aab295cbc1afd56b9ea6ced96e0717304/README.md","version":"3d4a294aab295cbc1afd56b9ea6ced96e0717304"}} {"id":"evidence-reported-rnaret-2026-license","kind":"source","name":"DrBlackZJU/RNAret LICENSE","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"a2010f343487d3f7618affe54f789f5487602331c0a8d03f49e9a7c547cf0499","artifact_url":"https://raw.githubusercontent.com/DrBlackZJU/RNAret/40ddab25fc038ba2b96bc9b9f88216abe38b2f64/LICENSE","retrieved_at":"2026-09-16T19:54:22.459346+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/DrBlackZJU/RNAret/blob/40ddab25fc038ba2b96bc9b9f88216abe38b2f64/LICENSE","version":"40ddab25fc038ba2b96bc9b9f88216abe38b2f64"}} {"id":"evidence-reported-rnaret-2026-readme-md","kind":"source","name":"DrBlackZJU/RNAret README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"b04da58b8774a9db3c6d873af7d49142c039afee958ab6dcd2d2ff3b29d7ddae","artifact_url":"https://raw.githubusercontent.com/DrBlackZJU/RNAret/40ddab25fc038ba2b96bc9b9f88216abe38b2f64/README.md","retrieved_at":"2026-09-16T19:54:22.459346+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/DrBlackZJU/RNAret/blob/40ddab25fc038ba2b96bc9b9f88216abe38b2f64/README.md","version":"40ddab25fc038ba2b96bc9b9f88216abe38b2f64"}} {"id":"evidence-reported-scalr-2025-license","kind":"source","name":"infocusp/scaLR LICENSE","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"55860dd9f0c93456c17596a049cb34cded65afa3219888cb7d0a3eb18e86ff68","artifact_url":"https://raw.githubusercontent.com/infocusp/scaLR/b5f72ce8f9bd25cb90f4b6b3112a03288c788f2e/LICENSE","retrieved_at":"2026-09-16T19:54:22.478513+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/infocusp/scaLR/blob/b5f72ce8f9bd25cb90f4b6b3112a03288c788f2e/LICENSE","version":"b5f72ce8f9bd25cb90f4b6b3112a03288c788f2e"}} {"id":"evidence-reported-scalr-2025-readme-md","kind":"source","name":"infocusp/scaLR README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"ba8ca22790d5265355298d4ca2adc7b5a01290d5bd07e45fac332b7dc57f31f5","artifact_url":"https://raw.githubusercontent.com/infocusp/scaLR/b5f72ce8f9bd25cb90f4b6b3112a03288c788f2e/README.md","retrieved_at":"2026-09-16T19:54:22.478513+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/infocusp/scaLR/blob/b5f72ce8f9bd25cb90f4b6b3112a03288c788f2e/README.md","version":"b5f72ce8f9bd25cb90f4b6b3112a03288c788f2e"}} {"id":"evidence-reported-scatac-llmda-2026-readme-md","kind":"source","name":"sheng-guan-2001/scLLMDA README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"b8327f7bbd867d6e5b7ab369873b59aeba64fe4ca79c879dd54c3b036b127d3b","artifact_url":"https://raw.githubusercontent.com/sheng-guan-2001/scLLMDA/5e24025710bb068312d50a5749ef6bb838ef5a32/README.md","retrieved_at":"2026-09-16T19:54:22.541360+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/sheng-guan-2001/scLLMDA/blob/5e24025710bb068312d50a5749ef6bb838ef5a32/README.md","version":"5e24025710bb068312d50a5749ef6bb838ef5a32"}} {"id":"evidence-reported-scregnet-2025-readme-md","kind":"source","name":"sindhura-cs/scRegNet README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"fa82e7aab8e650e488e267fbed583735e40f9d7e005e72d252c28860dd970d94","artifact_url":"https://raw.githubusercontent.com/sindhura-cs/scRegNet/30d0215efd99c40ceaceb37d161fc0a86a236e0b/README.md","retrieved_at":"2026-09-16T19:54:23.045099+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/sindhura-cs/scRegNet/blob/30d0215efd99c40ceaceb37d161fc0a86a236e0b/README.md","version":"30d0215efd99c40ceaceb37d161fc0a86a236e0b"}} {"id":"evidence-reported-scxdr-2026-readme-md","kind":"source","name":"QiGuan1920/scXDR2025 README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"e61ed3f85619ffed72cde43a6b3f13491f0788c37edf03266ffcbb77a0c3894c","artifact_url":"https://raw.githubusercontent.com/QiGuan1920/scXDR2025/5b39f37ba4df186eeea0881d59366458d2535db9/README.md","retrieved_at":"2026-09-16T19:54:23.123989+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/QiGuan1920/scXDR2025/blob/5b39f37ba4df186eeea0881d59366458d2535db9/README.md","version":"5b39f37ba4df186eeea0881d59366458d2535db9"}} {"id":"evidence-reported-single-cell-aging-probes-2026-readme-md","kind":"source","name":"Biodyn-AI/longevity-mechinterp README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"d858613c397dcfd9904d090eea810941babb7a98d6458275d71f7b8bc9689e71","artifact_url":"https://raw.githubusercontent.com/Biodyn-AI/longevity-mechinterp/5a61464632a3c3e8bebd396eb1ab17bce1dc2493/README.md","retrieved_at":"2026-09-16T19:54:23.526549+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/Biodyn-AI/longevity-mechinterp/blob/5a61464632a3c3e8bebd396eb1ab17bce1dc2493/README.md","version":"5a61464632a3c3e8bebd396eb1ab17bce1dc2493"}} {"id":"evidence-reported-structure-informed-current-html","kind":"source","name":"Structure-Informed Protein Language Models are Robust Predictors for Variant Effects (reviewed HTML snapshot)","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"6686d2646e7b1b1203a7ef6d48bacf966b4cd6b0895fd2855551937c6bb87111","artifact_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12068927/","retrieved_at":"2026-09-16T19:58:11.209175+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12068927/","version":"Human Genetics 2025 journal article (online 2024)"}} {"id":"evidence-reported-structure-informed-plm-2025-license","kind":"source","name":"Shen-Lab/Structure-informed_PLM LICENSE","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"4a8d4e3f2049f4ab2372435182b3cf913e9ee1c0c6efa56202f637017b94af50","artifact_url":"https://raw.githubusercontent.com/Shen-Lab/Structure-informed_PLM/2307b101f9bf08223729a68f52b8a6fb21f18991/LICENSE","retrieved_at":"2026-09-16T20:43:16.171470+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/Shen-Lab/Structure-informed_PLM/blob/2307b101f9bf08223729a68f52b8a6fb21f18991/LICENSE","version":"2307b101f9bf08223729a68f52b8a6fb21f18991"}} {"id":"evidence-reported-structure-informed-plm-2025-readme-md","kind":"source","name":"Shen-Lab/Structure-informed_PLM readMe.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"06787a14ac0e9a094fe524472bb2b06b5025aa8af122d913f3ed9c352938b161","artifact_url":"https://raw.githubusercontent.com/Shen-Lab/Structure-informed_PLM/2307b101f9bf08223729a68f52b8a6fb21f18991/readMe.md","retrieved_at":"2026-09-16T20:43:16.171470+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/Shen-Lab/Structure-informed_PLM/blob/2307b101f9bf08223729a68f52b8a6fb21f18991/readMe.md","version":"2307b101f9bf08223729a68f52b8a6fb21f18991"}} {"id":"evidence-reported-transbind-2026-license","kind":"source","name":"jianlin-cheng/TransBind LICENSE","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"3972dc9744f6499f0f9b2dbf76696f2ae7ad8af9b23dde66d6af86c9dfb36986","artifact_url":"https://raw.githubusercontent.com/jianlin-cheng/TransBind/7537f264c5ad94958bcad05bb57edd8028c323ff/LICENSE","retrieved_at":"2026-09-16T19:54:23.967114+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/jianlin-cheng/TransBind/blob/7537f264c5ad94958bcad05bb57edd8028c323ff/LICENSE","version":"7537f264c5ad94958bcad05bb57edd8028c323ff"}} {"id":"evidence-reported-transbind-2026-readme-md","kind":"source","name":"jianlin-cheng/TransBind README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"2cdf3d45d6b98e44131007ede55fd1bd5a599a4d1963879eac4f2a9a6e598af7","artifact_url":"https://raw.githubusercontent.com/jianlin-cheng/TransBind/7537f264c5ad94958bcad05bb57edd8028c323ff/README.md","retrieved_at":"2026-09-16T19:54:23.967114+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/jianlin-cheng/TransBind/blob/7537f264c5ad94958bcad05bb57edd8028c323ff/README.md","version":"7537f264c5ad94958bcad05bb57edd8028c323ff"}} {"id":"evidence-reported-tu-fold-2025-license","kind":"source","name":"ygjiyn/tu_fold LICENSE","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"69bd0ef33acdd9d4340c3925f27ef35141c26b4653b39deac3d78e5f5d5930c9","artifact_url":"https://raw.githubusercontent.com/ygjiyn/tu_fold/f0532b6bf38b2f57baf0ba6afce7766bfc64899b/LICENSE","retrieved_at":"2026-09-16T19:54:24.127143+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/ygjiyn/tu_fold/blob/f0532b6bf38b2f57baf0ba6afce7766bfc64899b/LICENSE","version":"f0532b6bf38b2f57baf0ba6afce7766bfc64899b"}} {"id":"evidence-reported-tu-fold-2025-readme-md","kind":"source","name":"ygjiyn/tu_fold README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"84d9bbc42e28b17e10a672d9a9539bf26988d92fee8287d9b186e11678c4824b","artifact_url":"https://raw.githubusercontent.com/ygjiyn/tu_fold/f0532b6bf38b2f57baf0ba6afce7766bfc64899b/README.md","retrieved_at":"2026-09-16T19:54:24.127143+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/ygjiyn/tu_fold/blob/f0532b6bf38b2f57baf0ba6afce7766bfc64899b/README.md","version":"f0532b6bf38b2f57baf0ba6afce7766bfc64899b"}} {"id":"evidence-reported-viral-contig-simulation-2021-license","kind":"source","name":"Strong-Lab/Viral_Classification_in_Metagenomics LICENSE","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"1b9591430c0d9b6e1beeb740ad2128a99f850a222ce85827a8070fdd54c7b286","artifact_url":"https://raw.githubusercontent.com/Strong-Lab/Viral_Classification_in_Metagenomics/f583cbff6b022ce3a7e3870003e22e14769566fa/LICENSE","retrieved_at":"2026-09-16T20:30:16.571343+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/Strong-Lab/Viral_Classification_in_Metagenomics/blob/f583cbff6b022ce3a7e3870003e22e14769566fa/LICENSE","version":"f583cbff6b022ce3a7e3870003e22e14769566fa"}} {"id":"evidence-reported-viral-contig-simulation-2021-readme-md","kind":"source","name":"Strong-Lab/Viral_Classification_in_Metagenomics README.md","description":"Primary-source artifact inspected during the configuration-profile evidence review. Its licence and scope are not inherited by unrelated models or data.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"eb2f64d6845419c0e643cbe03e9c15b28d2860bf4fe49a0a640b328d50b09f19","artifact_url":"https://raw.githubusercontent.com/Strong-Lab/Viral_Classification_in_Metagenomics/f583cbff6b022ce3a7e3870003e22e14769566fa/README.md","retrieved_at":"2026-09-16T20:30:16.571343+00:00","review_method":"automated_source_review","review_scope":"Explanatory metadata only; no new numerical result or independent reproduction.","url":"https://github.com/Strong-Lab/Viral_Classification_in_Metagenomics/blob/f583cbff6b022ce3a7e3870003e22e14769566fa/README.md","version":"f583cbff6b022ce3a7e3870003e22e14769566fa"}} {"id":"evidence-task-final-a-cammiq-2022-41467-2022-33869-moesm1-esm-pdf","kind":"source","name":"cammiq-2022__41467_2022_33869_MOESM1_ESM.pdf","description":"Primary-source artifact inspected for task-specific methodological metadata. The hash identifies the exact downloaded artifact or named member of its supplementary archive.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_member":"41467_2022_33869_MOESM1_ESM.pdf","artifact_sha256":"910aed130f3b4648b0758bdcc6b82d1e2d38ddb320ea80d670c98c565930610b","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC9616933/supplementaryFiles","retrieved_at":"2026-09-16T21:08:58.951180+00:00","review_method":"automated_source_review","review_scope":"Task metadata and methodological context; no numerical result changed or independently reproduced.","url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC9616933/supplementaryFiles","version":"Retrieved 2026-09-16; sha256:910aed130f3b4648b0758bdcc6b82d1e2d38ddb320ea80d670c98c565930610b"}} {"id":"evidence-task-final-a-cell-dino-2025-pcbi-1013828-s001-pdf","kind":"source","name":"cell-dino-2025__pcbi.1013828.s001.pdf","description":"Primary-source artifact inspected for task-specific methodological metadata. The hash identifies the exact downloaded artifact or named member of its supplementary archive.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_member":"pcbi.1013828.s001.pdf","artifact_sha256":"e19d3d5c8d9dc1abcaf699c571349d869a1a7fa8e293660f49f67819f4b4d283","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12826486/supplementaryFiles","retrieved_at":"2026-09-16T21:06:02.800324+00:00","review_method":"automated_source_review","review_scope":"Task metadata and methodological context; no numerical result changed or independently reproduced.","url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12826486/supplementaryFiles","version":"Retrieved 2026-09-16; sha256:e19d3d5c8d9dc1abcaf699c571349d869a1a7fa8e293660f49f67819f4b4d283"}} {"id":"evidence-task-final-a-clathrin-dataset-clathrin0-6-csv","kind":"source","name":"clathrin__Dataset__Clathrin0.6.csv","description":"Primary-source artifact inspected for task-specific methodological metadata. The hash identifies the exact downloaded artifact or named member of its supplementary archive.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"0c85f84ca65d8b4c2fbeaa92e33a2015ce3c7096bfa2b948d73c8716f8b8c739","artifact_url":"https://raw.githubusercontent.com/lawankorn-m/Clathrin/9a8f55bc008401180152864560d8c8528600fe71/Dataset/Clathrin0.6.csv","retrieved_at":"2026-09-16T21:12:58.536073+00:00","review_method":"automated_source_review","review_scope":"Task metadata and methodological context; no numerical result changed or independently reproduced.","url":"https://raw.githubusercontent.com/lawankorn-m/Clathrin/9a8f55bc008401180152864560d8c8528600fe71/Dataset/Clathrin0.6.csv","version":"9a8f55bc008401180152864560d8c8528600fe71"}} {"id":"evidence-task-final-a-clathrin-dataset-clathrin0-7-csv","kind":"source","name":"clathrin__Dataset__Clathrin0.7.csv","description":"Primary-source artifact inspected for task-specific methodological metadata. The hash identifies the exact downloaded artifact or named member of its supplementary archive.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"ba5444b11ed53cc1770e59b1e4b4a6078808ada83765edd418b644359d00184b","artifact_url":"https://raw.githubusercontent.com/lawankorn-m/Clathrin/9a8f55bc008401180152864560d8c8528600fe71/Dataset/Clathrin0.7.csv","retrieved_at":"2026-09-16T21:12:59.067575+00:00","review_method":"automated_source_review","review_scope":"Task metadata and methodological context; no numerical result changed or independently reproduced.","url":"https://raw.githubusercontent.com/lawankorn-m/Clathrin/9a8f55bc008401180152864560d8c8528600fe71/Dataset/Clathrin0.7.csv","version":"9a8f55bc008401180152864560d8c8528600fe71"}} {"id":"evidence-task-final-a-clathrin-dataset-clathrin1-0-csv","kind":"source","name":"clathrin__Dataset__Clathrin1.0.csv","description":"Primary-source artifact inspected for task-specific methodological metadata. The hash identifies the exact downloaded artifact or named member of its supplementary archive.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"260c306da68247a283e55c97b96de2750a98b459c91408b85d161f720cad6a79","artifact_url":"https://raw.githubusercontent.com/lawankorn-m/Clathrin/9a8f55bc008401180152864560d8c8528600fe71/Dataset/Clathrin1.0.csv","retrieved_at":"2026-09-16T21:12:59.571978+00:00","review_method":"automated_source_review","review_scope":"Task metadata and methodological context; no numerical result changed or independently reproduced.","url":"https://raw.githubusercontent.com/lawankorn-m/Clathrin/9a8f55bc008401180152864560d8c8528600fe71/Dataset/Clathrin1.0.csv","version":"9a8f55bc008401180152864560d8c8528600fe71"}} {"id":"evidence-task-final-a-clathrin-readme-md","kind":"source","name":"clathrin__README.md","description":"Primary-source artifact inspected for task-specific methodological metadata. The hash identifies the exact downloaded artifact or named member of its supplementary archive.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"341256ca55892a34c03617a6748f0b285f9b0bc4bc1d68c6c462c66efce59e50","artifact_url":"https://raw.githubusercontent.com/lawankorn-m/Clathrin/9a8f55bc008401180152864560d8c8528600fe71/README.md","retrieved_at":"2026-09-16T21:12:59.799010+00:00","review_method":"automated_source_review","review_scope":"Task metadata and methodological context; no numerical result changed or independently reproduced.","url":"https://raw.githubusercontent.com/lawankorn-m/Clathrin/9a8f55bc008401180152864560d8c8528600fe71/README.md","version":"9a8f55bc008401180152864560d8c8528600fe71"}} {"id":"evidence-task-final-a-deepinteraware-2025-advs-12-2412533-s001-pdf","kind":"source","name":"deepinteraware-2025__ADVS-12-2412533-s001.pdf","description":"Primary-source artifact inspected for task-specific methodological metadata. The hash identifies the exact downloaded artifact or named member of its supplementary archive.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_member":"ADVS-12-2412533-s001.pdf","artifact_sha256":"3c66d0d9561d744ca0e42a5d33c020719e6f46d7ed27e37a25c19c2bd8d89345","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11967782/supplementaryFiles","retrieved_at":"2026-09-16T21:05:56.618962+00:00","review_method":"automated_source_review","review_scope":"Task metadata and methodological context; no numerical result changed or independently reproduced.","url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11967782/supplementaryFiles","version":"Retrieved 2026-09-16; sha256:3c66d0d9561d744ca0e42a5d33c020719e6f46d7ed27e37a25c19c2bd8d89345"}} {"id":"evidence-task-final-a-enbed-2024-vbae117-supplementary-data-pdf","kind":"source","name":"enbed-2024__vbae117_supplementary_data.pdf","description":"Primary-source artifact inspected for task-specific methodological metadata. The hash identifies the exact downloaded artifact or named member of its supplementary archive.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_member":"vbae117_supplementary_data.pdf","artifact_sha256":"a169bbebba2298e9c98c33c28053c1a0b42d0c3e88e5a8795934725c4408cc32","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11341122/supplementaryFiles","retrieved_at":"2026-09-16T21:05:56.086158+00:00","review_method":"automated_source_review","review_scope":"Task metadata and methodological context; no numerical result changed or independently reproduced.","url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11341122/supplementaryFiles","version":"Retrieved 2026-09-16; sha256:a169bbebba2298e9c98c33c28053c1a0b42d0c3e88e5a8795934725c4408cc32"}} {"id":"evidence-task-final-a-gse108394-soft","kind":"source","name":"GSE108394.soft","description":"Primary-source artifact inspected for task-specific methodological metadata. The hash identifies the exact downloaded artifact or named member of its supplementary archive.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"e4b44c1b50d2a0e7a764630ea901d7848c72dc55cc66e9faf8b706154719fc17","artifact_url":"https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE108394&targ=self&form=text&view=quick","retrieved_at":"2026-09-16T21:11:57.422218+00:00","review_method":"automated_source_review","review_scope":"Task metadata and methodological context; no numerical result changed or independently reproduced.","url":"https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE108394&targ=self&form=text&view=quick","version":"Retrieved 2026-09-16; sha256:e4b44c1b50d2a0e7a764630ea901d7848c72dc55cc66e9faf8b706154719fc17"}} {"id":"evidence-task-final-a-gse117872-soft","kind":"source","name":"GSE117872.soft","description":"Primary-source artifact inspected for task-specific methodological metadata. The hash identifies the exact downloaded artifact or named member of its supplementary archive.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"f615587bbf6e89a29811400538966cdf008dc4a7719091fc44448ba4a4aa8f5c","artifact_url":"https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE117872&targ=self&form=text&view=quick","retrieved_at":"2026-09-16T21:11:57.754129+00:00","review_method":"automated_source_review","review_scope":"Task metadata and methodological context; no numerical result changed or independently reproduced.","url":"https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE117872&targ=self&form=text&view=quick","version":"Retrieved 2026-09-16; sha256:f615587bbf6e89a29811400538966cdf008dc4a7719091fc44448ba4a4aa8f5c"}} {"id":"evidence-task-final-a-gse127298-soft","kind":"source","name":"GSE127298.soft","description":"Primary-source artifact inspected for task-specific methodological metadata. The hash identifies the exact downloaded artifact or named member of its supplementary archive.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"40a50eb13981d4d9ca3dfcd987c6b4b90289385e6aaeea645c5672a2dc6a9f19","artifact_url":"https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE127298&targ=self&form=text&view=quick","retrieved_at":"2026-09-16T21:11:57.862686+00:00","review_method":"automated_source_review","review_scope":"Task metadata and methodological context; no numerical result changed or independently reproduced.","url":"https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE127298&targ=self&form=text&view=quick","version":"Retrieved 2026-09-16; sha256:40a50eb13981d4d9ca3dfcd987c6b4b90289385e6aaeea645c5672a2dc6a9f19"}} {"id":"evidence-task-final-a-gse134839-soft","kind":"source","name":"GSE134839.soft","description":"Primary-source artifact inspected for task-specific methodological metadata. The hash identifies the exact downloaded artifact or named member of its supplementary archive.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"333ba9b2538c6ba1f41ae08d60b41099f26a443d2b473cbd10104d887ef20064","artifact_url":"https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE134839&targ=self&form=text&view=quick","retrieved_at":"2026-09-16T21:11:57.450992+00:00","review_method":"automated_source_review","review_scope":"Task metadata and methodological context; no numerical result changed or independently reproduced.","url":"https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE134839&targ=self&form=text&view=quick","version":"Retrieved 2026-09-16; sha256:333ba9b2538c6ba1f41ae08d60b41099f26a443d2b473cbd10104d887ef20064"}} {"id":"evidence-task-final-a-gse140440-soft","kind":"source","name":"GSE140440.soft","description":"Primary-source artifact inspected for task-specific methodological metadata. The hash identifies the exact downloaded artifact or named member of its supplementary archive.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"7b6ca08ec013e2e79b367727f6ce4ff46dfb7ef9aa166bc4907dd02fbbb472e5","artifact_url":"https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE140440&targ=self&form=text&view=quick","retrieved_at":"2026-09-16T21:11:57.808765+00:00","review_method":"automated_source_review","review_scope":"Task metadata and methodological context; no numerical result changed or independently reproduced.","url":"https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE140440&targ=self&form=text&view=quick","version":"Retrieved 2026-09-16; sha256:7b6ca08ec013e2e79b367727f6ce4ff46dfb7ef9aa166bc4907dd02fbbb472e5"}} {"id":"evidence-task-final-a-gse147326-soft","kind":"source","name":"GSE147326.soft","description":"Primary-source artifact inspected for task-specific methodological metadata. The hash identifies the exact downloaded artifact or named member of its supplementary archive.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"6882efe837a56064645705857c81f21e6f82b773a8a7fd9679180f71bba9bb74","artifact_url":"https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE147326&targ=self&form=text&view=quick","retrieved_at":"2026-09-16T21:11:57.732001+00:00","review_method":"automated_source_review","review_scope":"Task metadata and methodological context; no numerical result changed or independently reproduced.","url":"https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE147326&targ=self&form=text&view=quick","version":"Retrieved 2026-09-16; sha256:6882efe837a56064645705857c81f21e6f82b773a8a7fd9679180f71bba9bb74"}} {"id":"evidence-task-final-a-gse149214-soft","kind":"source","name":"GSE149214.soft","description":"Primary-source artifact inspected for task-specific methodological metadata. The hash identifies the exact downloaded artifact or named member of its supplementary archive.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"801532dfc1983e4de5b4183d39e81f74825fa0b22632bb79322524d3f6139d1e","artifact_url":"https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE149214&targ=self&form=text&view=quick","retrieved_at":"2026-09-16T21:11:57.291347+00:00","review_method":"automated_source_review","review_scope":"Task metadata and methodological context; no numerical result changed or independently reproduced.","url":"https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE149214&targ=self&form=text&view=quick","version":"Retrieved 2026-09-16; sha256:801532dfc1983e4de5b4183d39e81f74825fa0b22632bb79322524d3f6139d1e"}} {"id":"evidence-task-final-a-gse164614-soft","kind":"source","name":"GSE164614.soft","description":"Primary-source artifact inspected for task-specific methodological metadata. The hash identifies the exact downloaded artifact or named member of its supplementary archive.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"132e5ca2603b5e5d91c16e708f8928245086d43d13f8f56ec5512373c4248e2c","artifact_url":"https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE164614&targ=self&form=text&view=quick","retrieved_at":"2026-09-16T21:11:57.455257+00:00","review_method":"automated_source_review","review_scope":"Task metadata and methodological context; no numerical result changed or independently reproduced.","url":"https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE164614&targ=self&form=text&view=quick","version":"Retrieved 2026-09-16; sha256:132e5ca2603b5e5d91c16e708f8928245086d43d13f8f56ec5512373c4248e2c"}} {"id":"evidence-task-final-a-gse230538-soft","kind":"source","name":"GSE230538.soft","description":"Primary-source artifact inspected for task-specific methodological metadata. The hash identifies the exact downloaded artifact or named member of its supplementary archive.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"63d3b23a3c1266d5cdbd04ca1f4766774ab8427e70b41dbc80e230275286cf0d","artifact_url":"https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE230538&targ=self&form=text&view=quick","retrieved_at":"2026-09-16T21:11:57.435478+00:00","review_method":"automated_source_review","review_scope":"Task metadata and methodological context; no numerical result changed or independently reproduced.","url":"https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE230538&targ=self&form=text&view=quick","version":"Retrieved 2026-09-16; sha256:63d3b23a3c1266d5cdbd04ca1f4766774ab8427e70b41dbc80e230275286cf0d"}} {"id":"evidence-task-final-a-hi-enhancer-2025-btaf441-supplementary-data-docx","kind":"source","name":"hi-enhancer-2025__btaf441_supplementary_data.docx","description":"Primary-source artifact inspected for task-specific methodological metadata. The hash identifies the exact downloaded artifact or named member of its supplementary archive.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_member":"btaf441_supplementary_data.docx","artifact_sha256":"26de5de88996a9da7723ba4036eb4cf234a77fae3bbab833edffbc7d99ec56a5","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12758598/supplementaryFiles","retrieved_at":"2026-09-16T21:05:56.618187+00:00","review_method":"automated_source_review","review_scope":"Task metadata and methodological context; no numerical result changed or independently reproduced.","url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12758598/supplementaryFiles","version":"Retrieved 2026-09-16; sha256:26de5de88996a9da7723ba4036eb4cf234a77fae3bbab833edffbc7d99ec56a5"}} {"id":"evidence-task-final-a-mrna-lm-2025-gkaf044-supplemental-file-pdf","kind":"source","name":"mrna-lm-2025__gkaf044_Supplemental_File.pdf","description":"Primary-source artifact inspected for task-specific methodological metadata. The hash identifies the exact downloaded artifact or named member of its supplementary archive.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_member":"gkaf044_Supplemental_File.pdf","artifact_sha256":"bf1e156bb09c90a9e18101a45332e5a7a8f3b7fd363f6ffbb7760020b4c8f40f","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11962594/supplementaryFiles","retrieved_at":"2026-09-16T21:08:59.937487+00:00","review_method":"automated_source_review","review_scope":"Task metadata and methodological context; no numerical result changed or independently reproduced.","url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11962594/supplementaryFiles","version":"Retrieved 2026-09-16; sha256:bf1e156bb09c90a9e18101a45332e5a7a8f3b7fd363f6ffbb7760020b4c8f40f"}} {"id":"evidence-task-final-a-plantcad2-2025-media-1-xlsx","kind":"source","name":"plantcad2-2025__media-1.xlsx","description":"Primary-source artifact inspected for task-specific methodological metadata. The hash identifies the exact downloaded artifact or named member of its supplementary archive.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_member":"media-1.xlsx","artifact_sha256":"f551e80bf044a0ea8ccc2e6f443fdf3690ca5e4a54d0e4bb4b2e6470ffc944be","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12425018/supplementaryFiles","retrieved_at":"2026-09-16T21:05:57.965868+00:00","review_method":"automated_source_review","review_scope":"Task metadata and methodological context; no numerical result changed or independently reproduced.","url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12425018/supplementaryFiles","version":"Retrieved 2026-09-16; sha256:f551e80bf044a0ea8ccc2e6f443fdf3690ca5e4a54d0e4bb4b2e6470ffc944be"}} {"id":"evidence-task-final-a-pmc6731122-xml","kind":"source","name":"PMC6731122.xml","description":"Primary-source artifact inspected for task-specific methodological metadata. The hash identifies the exact downloaded artifact or named member of its supplementary archive.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"2136d6c7589d2573e33c69f2b1eb0f2b76c696da06ee1c14427ccbaa45a33d5f","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC6731122/fullTextXML","retrieved_at":"2026-09-16T21:08:48.254469+00:00","review_method":"automated_source_review","review_scope":"Task metadata and methodological context; no numerical result changed or independently reproduced.","url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC6731122/fullTextXML","version":"Retrieved 2026-09-16; sha256:2136d6c7589d2573e33c69f2b1eb0f2b76c696da06ee1c14427ccbaa45a33d5f"}} {"id":"evidence-task-final-a-pmc7331607-xml","kind":"source","name":"PMC7331607.xml","description":"Primary-source artifact inspected for task-specific methodological metadata. The hash identifies the exact downloaded artifact or named member of its supplementary archive.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"d87b7b7a67445c04db4f1fbb0cc52914b1bccd04a23d4ff6d66e2be5373291c6","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC7331607/fullTextXML","retrieved_at":"2026-09-16T21:08:47.387344+00:00","review_method":"automated_source_review","review_scope":"Task metadata and methodological context; no numerical result changed or independently reproduced.","url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC7331607/fullTextXML","version":"Retrieved 2026-09-16; sha256:d87b7b7a67445c04db4f1fbb0cc52914b1bccd04a23d4ff6d66e2be5373291c6"}} {"id":"evidence-task-final-a-pmc7912887-xml","kind":"source","name":"PMC7912887.xml","description":"Primary-source artifact inspected for task-specific methodological metadata. The hash identifies the exact downloaded artifact or named member of its supplementary archive.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"3504a0fa3b62698b7eff78418e682bf2484236ed05645597a8be4fa2e97b3efc","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC7912887/fullTextXML","retrieved_at":"2026-09-16T21:08:40.609030+00:00","review_method":"automated_source_review","review_scope":"Task metadata and methodological context; no numerical result changed or independently reproduced.","url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC7912887/fullTextXML","version":"Retrieved 2026-09-16; sha256:3504a0fa3b62698b7eff78418e682bf2484236ed05645597a8be4fa2e97b3efc"}} {"id":"evidence-task-final-a-pmc9556750-xml","kind":"source","name":"PMC9556750.xml","description":"Primary-source artifact inspected for task-specific methodological metadata. The hash identifies the exact downloaded artifact or named member of its supplementary archive.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_sha256":"71fd84bba280e9ac7b1009600432245122e73f6a01916a743ef4faeb56bb26f0","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC9556750/fullTextXML","retrieved_at":"2026-09-16T21:08:44.485140+00:00","review_method":"automated_source_review","review_scope":"Task metadata and methodological context; no numerical result changed or independently reproduced.","url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC9556750/fullTextXML","version":"Retrieved 2026-09-16; sha256:71fd84bba280e9ac7b1009600432245122e73f6a01916a743ef4faeb56bb26f0"}} {"id":"evidence-task-final-a-rnaret-2026-42003-2026-9757-moesm2-esm-pdf","kind":"source","name":"rnaret-2026__42003_2026_9757_MOESM2_ESM.pdf","description":"Primary-source artifact inspected for task-specific methodological metadata. The hash identifies the exact downloaded artifact or named member of its supplementary archive.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_member":"42003_2026_9757_MOESM2_ESM.pdf","artifact_sha256":"8464ff052a60b946bd08fa18860f22b2f390fe93dd0549eb845c8e9a68cf5fcc","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13111708/supplementaryFiles","retrieved_at":"2026-09-16T21:06:05.136898+00:00","review_method":"automated_source_review","review_scope":"Task metadata and methodological context; no numerical result changed or independently reproduced.","url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13111708/supplementaryFiles","version":"Retrieved 2026-09-16; sha256:8464ff052a60b946bd08fa18860f22b2f390fe93dd0549eb845c8e9a68cf5fcc"}} {"id":"evidence-task-final-a-scxdr-2026-42003-2025-9418-moesm1-esm-pdf","kind":"source","name":"scxdr-2026__42003_2025_9418_MOESM1_ESM.pdf","description":"Primary-source artifact inspected for task-specific methodological metadata. The hash identifies the exact downloaded artifact or named member of its supplementary archive.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"artifact_member":"42003_2025_9418_MOESM1_ESM.pdf","artifact_sha256":"ad53da81235ba47f762c93ac5140108241b857160d978c63ff934e8e57283758","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12859067/supplementaryFiles","retrieved_at":"2026-09-16T21:05:58.741824+00:00","review_method":"automated_source_review","review_scope":"Task metadata and methodological context; no numerical result changed or independently reproduced.","url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12859067/supplementaryFiles","version":"Retrieved 2026-09-16; sha256:ad53da81235ba47f762c93ac5140108241b857160d978c63ff934e8e57283758"}} {"id":"evidence-task-final-b-birna-supplement","kind":"source","name":"birna journal Supplementary Information — pinned PDF","description":"Published supplementary PDF inspected for task-specific computational evaluation metadata.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12635123/supplementaryFiles","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12635123/supplementaryFiles","artifact_sha256":"7897d4dcf456d1f22b0631beabf7c5fd8678d0b8765cd40325195eb5addf4493","archive_member":"42003_2025_8982_MOESM2_ESM.pdf","archive_sha256":"f30f2e52b0ef64680b9a544b9db444d32ef302b7d9e09682b1ca339ae9086d23","version":"Published supplementary PDF 42003_2025_8982_MOESM2_ESM.pdf; sha256:7897d4dcf456d1f22b0631beabf7c5fd8678d0b8765cd40325195eb5addf4493","retrieved_at":"2026-09-16T21:06:10.816352+00:00","review_scope":"Task-specific evaluation passages and reporting scope checked; no numerical results reproduced."}} {"id":"evidence-task-final-b-mrnabert-supplement","kind":"source","name":"mrnabert journal Supplementary Information — pinned PDF","description":"Published supplementary PDF inspected for task-specific computational evaluation metadata.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12644827/supplementaryFiles","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12644827/supplementaryFiles","artifact_sha256":"2820d9385389e41e9213223608b26c84bf6bffbba57ff48fd5dda17fb78a8aed","archive_member":"41467_2025_65340_MOESM1_ESM.pdf","archive_sha256":"c89fe42941b7cc5a4ec7007ba236d500c3a14d4c925ac6e92d58be1307666538","version":"Published supplementary PDF 41467_2025_65340_MOESM1_ESM.pdf; sha256:2820d9385389e41e9213223608b26c84bf6bffbba57ff48fd5dda17fb78a8aed","retrieved_at":"2026-09-16T21:06:12.558631+00:00","review_scope":"Task-specific evaluation passages and reporting scope checked; no numerical results reproduced."}} {"id":"evidence-task-final-b-pst-supplement","kind":"source","name":"pst journal Supplementary Information — pinned PDF","description":"Published supplementary PDF inspected for task-specific computational evaluation metadata.","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12603367/supplementaryFiles","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12603367/supplementaryFiles","artifact_sha256":"f8c3375644f10aeb50c76f99071ec64b2b0dceadcee7f70a809adbcc8985d936","archive_member":"btaf582_supplementary_data.pdf","archive_sha256":"6fbc2e0c0f3a5d5128bd7a2f968b838de3485ce1ba6b090694c130c3e2051549","version":"Published supplementary PDF btaf582_supplementary_data.pdf; sha256:f8c3375644f10aeb50c76f99071ec64b2b0dceadcee7f70a809adbcc8985d936","retrieved_at":"2026-09-16T21:06:10.146345+00:00","review_scope":"Task-specific evaluation passages and reporting scope checked; no numerical results reproduced."}} {"id":"fingerprint-scoring-2022","kind":"source","name":"Machine-Learning- and Knowledge-Based Scoring Functions Incorporating Ligand and Protein Fingerprints","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC9178954/","version":"PMC archival version PMC9178954.1","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.1021/acsomega.2c02822","publication_status":"peer_reviewed","year":2022,"artifact_sha256":"47bd60c6392b801095fdb604de06c0d4bda6f555bae58e9955e491a5abf60576","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC9178954/fullTextXML","artifact_retrieved_at":"2026-09-16T10:44:03.439460+00:00","legacy_paper":{"id":"fingerprint-scoring-2022","title":"Machine-Learning- and Knowledge-Based Scoring Functions Incorporating Ligand and Protein Fingerprints","year":2022,"publication_status":"peer_reviewed","version":"PMC archival version PMC9178954.1","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC9178954/","primary_domain":"molecular-interactions","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: ACS Omega; PMC ID: PMC9178954.","doi":"10.1021/acsomega.2c02822"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"fujisan-2024","kind":"source","name":"Enhanced prediction of protein functional identity through the integration of sequence and structural features","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["proteins-complexes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11609699/","version":"PMC11609699.1","retrieved_at":"2026-09-15T23:25:00Z","doi":"10.1016/j.csbj.2024.11.028","publication_status":"peer_reviewed","year":2024,"artifact_sha256":"db33e0542005ffae00cd644dfe697185b94c8823d5aee2768620a6db0c48e56f","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11609699/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:35.728Z","legacy_paper":{"id":"fujisan-2024","title":"Enhanced prediction of protein functional identity through the integration of sequence and structural features","year":2024,"publication_status":"peer_reviewed","version":"PMC11609699.1","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11609699/","primary_domain":"proteins-complexes","retrieved_utc":"2026-09-15T23:25:00Z","notes":"Primary full text via Europe PMC XML; venue: Computational and Structural Biotechnology Journal; PMC ID: PMC11609699.","doi":"10.1016/j.csbj.2024.11.028"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"fusion-breakpoint-foundation-models-2026","kind":"source","name":"Benchmarking genomic foundation models for binary classification of gene fusion breakpoints from DNA sequences","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["dna-genomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13182013/","version":"journal full text in PMC","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.1186/s13040-026-00553-1","publication_status":"peer_reviewed","year":2026,"artifact_sha256":"0f4d9de77f1e39cfd2164a20653d86370767da684dc22d17e09f589761abeb5f","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13182013/fullTextXML","artifact_retrieved_at":"2026-09-16T10:38:57.558209+00:00","legacy_paper":{"id":"fusion-breakpoint-foundation-models-2026","title":"Benchmarking genomic foundation models for binary classification of gene fusion breakpoints from DNA sequences","year":2026,"publication_status":"peer_reviewed","version":"journal full text in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13182013/","primary_domain":"dna-genomes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.1186/s13040-026-00553-1","notes":"Numeric result checked against Table 2 in primary full-text XML; journal/source: BioData Mining."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"genept-2024","kind":"source","name":"GenePT: A Simple But Effective Foundation Model for Genes and Cells Built From ChatGPT","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["cells-tissues"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC10614824/","version":"PMC archival version PMC10614824.2","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.1101/2023.10.16.562533","publication_status":"preprint","year":2024,"artifact_sha256":"230a2ec55458d9243eaeeebf3244df7409eb02d47f4b809ee56a06dcb6fdd047","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC10614824/fullTextXML","artifact_retrieved_at":"2026-09-16T10:44:03.399274+00:00","legacy_paper":{"id":"genept-2024","title":"GenePT: A Simple But Effective Foundation Model for Genes and Cells Built From ChatGPT","year":2024,"publication_status":"preprint","version":"PMC archival version PMC10614824.2","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC10614824/","primary_domain":"cells-tissues","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: bioRxiv; PMC ID: PMC10614824.","doi":"10.1101/2023.10.16.562533"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"genomeocean-2025","kind":"source","name":"GenomeOcean: An Efficient Genome Foundation Model Trained on Large-Scale Metagenomic Assemblies","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11838515/","version":"preprint archived 2025-02-05","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.1101/2025.01.30.635558","publication_status":"preprint","year":2025,"artifact_sha256":"3cc0df52522fccda23e3958f069c916b87ee50bb5c9a992fa37e25256546e145","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11838515/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:55.224Z","legacy_paper":{"id":"genomeocean-2025","title":"GenomeOcean: An Efficient Genome Foundation Model Trained on Large-Scale Metagenomic Assemblies","year":2025,"publication_status":"preprint","version":"preprint archived 2025-02-05","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11838515/","primary_domain":"microbes-communities","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: bioRxiv; PMC ID: PMC11838515.","doi":"10.1101/2025.01.30.635558"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"genomic-tokenizer-selection-2025","kind":"source","name":"The impact of tokenizer selection in genomic language models","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["dna-genomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12453675/","version":"journal full text in PMC","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.1093/bioinformatics/btaf456","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"0a01c36fdd63f3f6db509777e61c3f87e8a298c810f8aef7974915aaa0655342","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12453675/fullTextXML","artifact_retrieved_at":"2026-09-16T10:38:57.558210+00:00","legacy_paper":{"id":"genomic-tokenizer-selection-2025","title":"The impact of tokenizer selection in genomic language models","year":2025,"publication_status":"peer_reviewed","version":"journal full text in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12453675/","primary_domain":"dna-genomes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.1093/bioinformatics/btaf456","notes":"Final Bioinformatics journal article Table 2, Caduceus (char) Regulatory MCC 0.778 checked directly; same study also has a bioRxiv manuscript."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"gremln-2026","kind":"source","name":"GREmLN: A Cellular Graph Structure Aware Transcriptomics Foundation Model","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["cells-tissues"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13060794/","version":"preprint version in PMC","retrieved_at":"2026-09-15T23:37:05Z","doi":"10.1101/2025.07.03.663009","publication_status":"preprint","year":2026,"artifact_sha256":"3a20c4ededb749fc3f1120baf16dcfebe3fcb30418a91c445cfd91a7b5fdf553","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13060794/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:57.502Z","legacy_paper":{"id":"gremln-2026","title":"GREmLN: A Cellular Graph Structure Aware Transcriptomics Foundation Model","year":2026,"publication_status":"preprint","version":"preprint version in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13060794/","primary_domain":"cells-tissues","retrieved_utc":"2026-09-15T23:37:05Z","notes":"Primary full text verified via Europe PMC fullTextXML; venue: bioRxiv; PMC ID: PMC13060794. Preprint; table labels metric F1; paper does not specify macro in this row.","doi":"10.1101/2025.07.03.663009"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"gsmformer-ppi-2026","kind":"source","name":"Multimodal graph, surface, and language-based model for protein protein interaction prediction","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["proteins-complexes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12873117/","version":"journal full text in PMC","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.1038/s41598-025-34758-x","publication_status":"peer_reviewed","year":2026,"artifact_sha256":"9b364b5d73d16f2787f93f78f17dbe98b954ab9c2c64c1df960eec2e615eb3b4","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12873117/fullTextXML","artifact_retrieved_at":"2026-09-16T10:38:57.558212+00:00","legacy_paper":{"id":"gsmformer-ppi-2026","title":"Multimodal graph, surface, and language-based model for protein protein interaction prediction","year":2026,"publication_status":"peer_reviewed","version":"journal full text in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12873117/","primary_domain":"proteins-complexes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.1038/s41598-025-34758-x","notes":"Numeric result checked against Table 6 in primary full-text XML; journal/source: Scientific Reports."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"hi-enhancer-2025","kind":"source","name":"Hi-Enhancer: a two-stage framework for prediction and localization of enhancers based on Blending-KAN and Stacking-Auto models","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["dna-genomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12758598/","version":"version of record","retrieved_at":"2026-09-15T23:37:05Z","doi":"10.1093/bioinformatics/btaf441","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"c86488c9f60329b7a3c4370598e7a0a9e4c8c45d1758b87007bfc8242376b009","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12758598/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:06Z","legacy_paper":{"id":"hi-enhancer-2025","title":"Hi-Enhancer: a two-stage framework for prediction and localization of enhancers based on Blending-KAN and Stacking-Auto models","year":2025,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12758598/","primary_domain":"dna-genomes","retrieved_utc":"2026-09-15T23:37:05Z","notes":"Primary full text verified via Europe PMC fullTextXML; venue: Bioinformatics; PMC ID: PMC12758598. Task-specific enhancer predictor; not a DNA foundation model. Comparison values from older papers excluded.","doi":"10.1093/bioinformatics/btaf441"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"ibex-2025","kind":"source","name":"Conformation-aware structure prediction of antigen-recognizing immune proteins","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12710905/","version":"PMC archival version PMC12710905.1","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.1080/19420862.2025.2602217","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"caa1109bd5fe7f6be703aa9d4afd6f4f1522bcbce6b7361650eb59618c2a9e14","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12710905/fullTextXML","artifact_retrieved_at":"2026-09-16T10:44:03.426811+00:00","legacy_paper":{"id":"ibex-2025","title":"Conformation-aware structure prediction of antigen-recognizing immune proteins","year":2025,"publication_status":"peer_reviewed","version":"PMC archival version PMC12710905.1","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12710905/","primary_domain":"molecular-interactions","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: mAbs; PMC ID: PMC12710905.","doi":"10.1080/19420862.2025.2602217"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"icctax-2025","kind":"source","name":"ICCTax: a hierarchical taxonomic classifier for metagenomic sequences on a large language model","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12619997/","version":"version of record","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.1093/bioadv/vbaf257","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"2ce0b48f1cde3aea7e561d92f4d7dc1525af7439ccd16f80bec0773e8812c8ec","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12619997/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:55.373Z","legacy_paper":{"id":"icctax-2025","title":"ICCTax: a hierarchical taxonomic classifier for metagenomic sequences on a large language model","year":2025,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12619997/","primary_domain":"microbes-communities","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: Bioinformatics Advances; PMC ID: PMC12619997.","doi":"10.1093/bioadv/vbaf257"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"insilico-perturbation-auprc-2025","kind":"source","name":"AUPRC: a metric for evaluating the performance of in-silico perturbation methods in identifying differentially expressed genes","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["cells-tissues"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12400816/","version":"PMC archival version PMC12400816.1","retrieved_at":"2026-09-15T23:37:05Z","doi":"10.1093/bib/bbaf426","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"2715709d94f84744afa32cafdcaa72efd206d63af8c60afe7619b2cb90108b6b","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12400816/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:06Z","legacy_paper":{"id":"insilico-perturbation-auprc-2025","title":"AUPRC: a metric for evaluating the performance of in-silico perturbation methods in identifying differentially expressed genes","year":2025,"publication_status":"peer_reviewed","version":"PMC archival version PMC12400816.1","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12400816/","primary_domain":"cells-tissues","retrieved_utc":"2026-09-15T23:37:05Z","notes":"Primary full text verified via Europe PMC fullTextXML; venue: Briefings in Bioinformatics; PMC ID: PMC12400816. Paper benchmarks metrics and scGen perturbation method; no foundation-model result in this row.","doi":"10.1093/bib/bbaf426"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"ipromp-2025","kind":"source","name":"iPro-MP: a BERT-based model to predict multiple prokaryotic promoters","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12516880/","version":"version of record","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.1186/s13059-025-03819-9","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"d21541ee1f7a168da8e4a7c0f0e133c970cbe7bc41118f43a929f08b2fd2afd1","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12516880/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:55.361Z","legacy_paper":{"id":"ipromp-2025","title":"iPro-MP: a BERT-based model to predict multiple prokaryotic promoters","year":2025,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12516880/","primary_domain":"microbes-communities","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: Genome Biology; PMC ID: PMC12516880.","doi":"10.1186/s13059-025-03819-9"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"kmetashot-2025","kind":"source","name":"kMetaShot: a fast and reliable taxonomy classifier for metagenome-assembled genomes","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11695915/","version":"PMC archival version PMC11695915.1","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.1093/bib/bbae680","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"4584e93ea035c1170b8756a0a52cbe99fe72e70bd09b5f1dee639ee104f78247","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11695915/fullTextXML","artifact_retrieved_at":"2026-09-16T10:44:03.417367+00:00","legacy_paper":{"id":"kmetashot-2025","title":"kMetaShot: a fast and reliable taxonomy classifier for metagenome-assembled genomes","year":2025,"publication_status":"peer_reviewed","version":"PMC archival version PMC11695915.1","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11695915/","primary_domain":"microbes-communities","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: Briefings in Bioinformatics; PMC ID: PMC11695915.","doi":"10.1093/bib/bbae680"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"lambda-prophage-2026","kind":"source","name":"LAMBDA: A Prophage Detection Benchmark for Genomic Language Models","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13041943/","version":"PMC13041943.1","retrieved_at":"2026-09-15T23:25:00Z","doi":"10.64898/2026.03.26.714501","publication_status":"preprint","year":2026,"artifact_sha256":"22c2e218e87dce757907f6086a0e2ad37c13f785b34fff5bea7cfa1a6c276b16","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13041943/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:36.240Z","legacy_paper":{"id":"lambda-prophage-2026","title":"LAMBDA: A Prophage Detection Benchmark for Genomic Language Models","year":2026,"publication_status":"preprint","version":"PMC13041943.1","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13041943/","primary_domain":"microbes-communities","retrieved_utc":"2026-09-15T23:25:00Z","notes":"Primary full text via Europe PMC XML; venue: bioRxiv; PMC ID: PMC13041943.","doi":"10.64898/2026.03.26.714501"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"lazypipe-2020","kind":"source","name":"Novel NGS pipeline for virus discovery from a wide spectrum of hosts and sample types","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC7772471/","version":"version of record","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.1093/ve/veaa091","publication_status":"peer_reviewed","year":2020,"artifact_sha256":"77842d8e4f6b419e331ab5a01fdf8f9eb8604f259425d79602be896aad3d0ad1","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC7772471/fullTextXML","artifact_retrieved_at":"2026-09-16T10:44:03.412605+00:00","legacy_paper":{"id":"lazypipe-2020","title":"Novel NGS pipeline for virus discovery from a wide spectrum of hosts and sample types","year":2020,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC7772471/","primary_domain":"microbes-communities","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: Virus Evolution; PMC ID: PMC7772471.","doi":"10.1093/ve/veaa091"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"lemur-magnet-2024","kind":"source","name":"Lightweight taxonomic profiling of long-read metagenomic datasets with Lemur and Magnet","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11185576/","version":"PMC archival version PMC11185576.2","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.1101/2024.06.01.596961","publication_status":"preprint","year":2024,"artifact_sha256":"4afb9195da447916eb6f733816e3640741c7ade08ea8920d205c3be7b3cce27a","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11185576/fullTextXML","artifact_retrieved_at":"2026-09-16T10:44:03.420493+00:00","legacy_paper":{"id":"lemur-magnet-2024","title":"Lightweight taxonomic profiling of long-read metagenomic datasets with Lemur and Magnet","year":2024,"publication_status":"preprint","version":"PMC archival version PMC11185576.2","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11185576/","primary_domain":"microbes-communities","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: bioRxiv; PMC ID: PMC11185576.","doi":"10.1101/2024.06.01.596961"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"ligand-affinity-meta-model-2024","kind":"source","name":"Improved Prediction of Ligand–Protein Binding Affinities by Meta-modeling","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11632770/","version":"PMC archival version PMC11632770.1","retrieved_at":"2026-09-15T23:37:05Z","doi":"10.1021/acs.jcim.4c01116","publication_status":"peer_reviewed","year":2024,"artifact_sha256":"0be25fe75bc0b2eb8065136555763bbae5964ea3a8fdb8c5de79ff96445f6a29","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11632770/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:06Z","legacy_paper":{"id":"ligand-affinity-meta-model-2024","title":"Improved Prediction of Ligand–Protein Binding Affinities by Meta-modeling","year":2024,"publication_status":"peer_reviewed","version":"PMC archival version PMC11632770.1","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11632770/","primary_domain":"molecular-interactions","retrieved_utc":"2026-09-15T23:37:05Z","notes":"Primary full text verified via Europe PMC fullTextXML; venue: Journal of Chemical Information and Modeling; PMC ID: PMC11632770. Mixed prediction units across Table 4 comparators; only meta-model PCC recorded.","doi":"10.1021/acs.jcim.4c01116"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"lipp-2026","kind":"source","name":"The LiPP Benchmark Set for Modeling Lipid–Protein Complexes: Comparison of Co-Folding and Docking Methods","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13292216/","version":"PMC13292216.1","retrieved_at":"2026-09-15T23:25:00Z","doi":"10.1021/acs.jcim.6c01457","publication_status":"peer_reviewed","year":2026,"artifact_sha256":"6ff34f2f709a14858a3753abf9f8f6efa1e7e3c351f15c70cf64264193a9414e","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13292216/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:16.548973+00:00","legacy_paper":{"id":"lipp-2026","title":"The LiPP Benchmark Set for Modeling Lipid–Protein Complexes: Comparison of Co-Folding and Docking Methods","year":2026,"publication_status":"peer_reviewed","version":"PMC13292216.1","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13292216/","primary_domain":"molecular-interactions","retrieved_utc":"2026-09-15T23:25:00Z","notes":"Primary full text via Europe PMC XML; venue: Journal of Chemical Information and Modeling; PMC ID: PMC13292216.","doi":"10.1021/acs.jcim.6c01457"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"lit-001","kind":"result","name":"Caduceus-Ph · AUC · Human 5mC","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["Human 5mC detection"]},"source_ids":["dna-foundation-models-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-001"}],"attributes":{"printed_value":"0.783","numeric_value":"0.783","metric":"AUC","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 3, Human 5mC row, Caduceus-Ph column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:35.327Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 3, Human 5mC row, Caduceus-Ph column; cell: 0.783","artifact_sha256":"5d8ca9bcf88cc1b38ad667906a2e4699b1aefa6d31c6f49259784930353f3202","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12663285/fullTextXML"},"legacy_id":"lit-001","legacy_row":{"id":"lit-001","paper_id":"dna-foundation-models-2025","domain_id":"dna-genomes","task":"Human 5mC detection","model":"Caduceus-Ph","model_version":"","dataset":"Human 5mC","dataset_version":"","split":"","metric":"AUC","value":"0.783","unit":"unitless","uncertainty":"","protocol":"Binary epigenetic-modification classification as reported in the paper.","source_locator":"Table 3, Human 5mC row, Caduceus-Ph column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12663285/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-002","kind":"result","name":"NT-v2 · AUC · Human 5mC","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["Human 5mC detection"]},"source_ids":["dna-foundation-models-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-002"}],"attributes":{"printed_value":"0.7377","numeric_value":"0.7377","metric":"AUC","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 3, Human 5mC row, NT-v2 column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:35.327Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 3, Human 5mC row, NT-v2 column; cell: 0.7377","artifact_sha256":"5d8ca9bcf88cc1b38ad667906a2e4699b1aefa6d31c6f49259784930353f3202","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12663285/fullTextXML"},"legacy_id":"lit-002","legacy_row":{"id":"lit-002","paper_id":"dna-foundation-models-2025","domain_id":"dna-genomes","task":"Human 5mC detection","model":"NT-v2","model_version":"","dataset":"Human 5mC","dataset_version":"","split":"","metric":"AUC","value":"0.7377","unit":"unitless","uncertainty":"","protocol":"Binary epigenetic-modification classification as reported in the paper.","source_locator":"Table 3, Human 5mC row, NT-v2 column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12663285/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-003","kind":"result","name":"ENBED · Accuracy · Genomic Benchmarks Mouse Enhancers","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["Enhancer classification"]},"source_ids":["enbed-2024"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-003"}],"attributes":{"printed_value":"90.3","numeric_value":"90.3","metric":"Accuracy","metric_direction":"unknown","unit":"%","uncertainty":null,"source_locator":"Table 2, Mouse Enhancers row, ENBED column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:35.378Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 2, Mouse Enhancers row, ENBED column; cell: 90.3","artifact_sha256":"e95d4be70d32e61af5a92eda8ea66f25a2cc629e3f83e7b5241b13cde8bdb83b","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11341122/fullTextXML"},"legacy_id":"lit-003","legacy_row":{"id":"lit-003","paper_id":"enbed-2024","domain_id":"dna-genomes","task":"Enhancer classification","model":"ENBED","model_version":"","dataset":"Genomic Benchmarks Mouse Enhancers","dataset_version":"","split":"","metric":"Accuracy","value":"90.3","unit":"%","uncertainty":"","protocol":"Reported Genomic Benchmarks classification accuracy.","source_locator":"Table 2, Mouse Enhancers row, ENBED column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11341122/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-004","kind":"result","name":"ENBED (GRCh38) · Accuracy · Genomic Benchmarks Mouse Enhancers","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["Enhancer classification"]},"source_ids":["enbed-2024"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-004"}],"attributes":{"printed_value":"81.1","numeric_value":"81.1","metric":"Accuracy","metric_direction":"unknown","unit":"%","uncertainty":null,"source_locator":"Table 2, Mouse Enhancers row, ENBED (GRCh38) column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:35.378Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 2, Mouse Enhancers row, ENBED (GRCh38) column; cell: 81.1","artifact_sha256":"e95d4be70d32e61af5a92eda8ea66f25a2cc629e3f83e7b5241b13cde8bdb83b","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11341122/fullTextXML"},"legacy_id":"lit-004","legacy_row":{"id":"lit-004","paper_id":"enbed-2024","domain_id":"dna-genomes","task":"Enhancer classification","model":"ENBED (GRCh38)","model_version":"","dataset":"Genomic Benchmarks Mouse Enhancers","dataset_version":"","split":"","metric":"Accuracy","value":"81.1","unit":"%","uncertainty":"","protocol":"ENBED trained on GRCh38; reported Genomic Benchmarks classification accuracy.","source_locator":"Table 2, Mouse Enhancers row, ENBED (GRCh38) column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11341122/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-005","kind":"result","name":"DNABERT-2 · Accuracy · KEx","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["G-quadruplex classification"]},"source_ids":["quadruplex-llm-benchmark-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-005"}],"attributes":{"printed_value":"97.0","numeric_value":"97.0","metric":"Accuracy","metric_direction":"unknown","unit":"%","uncertainty":"± 0.5","source_locator":"Table 5, DNABERT-2 (117 M) row, Accuracy column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:35.379Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 5, DNABERT-2 (117 M) row, Accuracy column; cell: 97.0 ± 0.5","artifact_sha256":"c3d7c6d068d3c11a9c8255a932197ece3804d94e8b2d4f858bea373a1b6eb32f","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11953744/fullTextXML"},"legacy_id":"lit-005","legacy_row":{"id":"lit-005","paper_id":"quadruplex-llm-benchmark-2025","domain_id":"dna-genomes","task":"G-quadruplex classification","model":"DNABERT-2","model_version":"117M","dataset":"KEx","dataset_version":"","split":"","metric":"Accuracy","value":"97.0","unit":"%","uncertainty":"± 0.5","protocol":"Pretrained model evaluated on KEx as reported in Table 5.","source_locator":"Table 5, DNABERT-2 (117 M) row, Accuracy column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11953744/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"lit-006","kind":"result","name":"Caduceus · Accuracy · KEx","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["G-quadruplex classification"]},"source_ids":["quadruplex-llm-benchmark-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-006"}],"attributes":{"printed_value":"95.0","numeric_value":"95.0","metric":"Accuracy","metric_direction":"unknown","unit":"%","uncertainty":"± 0.5","source_locator":"Table 5, Caduceus (8 M) row, Accuracy column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:35.379Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 5, Caduceus (8 M) row, Accuracy column; cell: 95.0 ± 0.5","artifact_sha256":"c3d7c6d068d3c11a9c8255a932197ece3804d94e8b2d4f858bea373a1b6eb32f","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11953744/fullTextXML"},"legacy_id":"lit-006","legacy_row":{"id":"lit-006","paper_id":"quadruplex-llm-benchmark-2025","domain_id":"dna-genomes","task":"G-quadruplex classification","model":"Caduceus","model_version":"8M","dataset":"KEx","dataset_version":"","split":"","metric":"Accuracy","value":"95.0","unit":"%","uncertainty":"± 0.5","protocol":"Pretrained model evaluated on KEx as reported in Table 5.","source_locator":"Table 5, Caduceus (8 M) row, Accuracy column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11953744/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"lit-007","kind":"result","name":"HyenaDNA · AUROC · DNALongBench ETGP","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["Enhancer-target gene prediction"]},"source_ids":["dnalongbench-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-007"}],"attributes":{"printed_value":"0.828","numeric_value":"0.828","metric":"AUROC","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 3, HyenaDNA row, ETGP column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.492545+00:00","notes":"ETGP is the first numeric column, separate from the six CMP cell-type columns and average; caption defines ETGP AUROC. This verifies the central score at its source location, not every metadata field or an experimental reproduction.","evidence":"{\"table_xml_id\": \"T3\", \"row_cells\": [\"HyenaDNA\", \"0.828\", \"0.139\", \"0.122\", \"0.099\", \"0.097\", \"0.118\", \"0.115\"], \"selected_cell_zero_based\": 1, \"selected_cell_xml\": \"